@tangle-network/agent-eval 0.161.1 → 0.170.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +156 -0
- package/README.md +2 -0
- package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
- package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
- package/dist/adapters/http.d.ts +108 -0
- package/dist/adapters/http.d.ts.map +1 -0
- package/dist/adapters/http.js +208 -0
- package/dist/adapters/http.js.map +1 -0
- package/dist/analyst/index.d.ts +40 -70
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +18 -311
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
- package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
- package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
- package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
- package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
- package/dist/benchmark-C4wk_Sjr.js.map +1 -0
- package/dist/{benchmark-command-BDC3Gocz.js → benchmark-command-BA7qOdWw.js} +239 -253
- package/dist/benchmark-command-BA7qOdWw.js.map +1 -0
- package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
- package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +22 -8
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BSmOwskD.js → campaign-BeCbxFqs.js} +19 -18
- package/dist/campaign-BeCbxFqs.js.map +1 -0
- package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
- package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
- package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
- package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
- package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
- package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
- package/dist/cli.js +3 -3
- package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
- package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
- package/dist/{client-L9VVPkim.d.ts → client-_Fsa5c2_.d.ts} +4 -4
- package/dist/{client-L9VVPkim.d.ts.map → client-_Fsa5c2_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -703
- package/dist/contract/index.js +14 -14
- package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
- package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
- package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
- package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
- package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-ovxrOP0_.d.ts} +6 -6
- package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-ovxrOP0_.d.ts.map} +1 -1
- package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-Clj-8igZ.js} +27 -15
- package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-Clj-8igZ.js.map} +1 -1
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-DVJm8Xlh.d.ts} +7 -7
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-DVJm8Xlh.d.ts.map} +1 -1
- package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
- package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
- package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
- package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
- package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
- package/dist/emitter-DeQHiDMm.js.map +1 -0
- package/dist/{engine-Cu5qD5Fc.d.ts → engine-D12Rb6WB.d.ts} +7 -7
- package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-D12Rb6WB.d.ts.map} +1 -1
- package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-JDTeE6Pl.js} +5 -5
- package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
- package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
- package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +5 -5
- package/dist/experiment/index.js +11 -11
- package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
- package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-CQxylYeG.js} +4 -11
- package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-Cex8Da2i.js} +26 -12
- package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
- package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
- package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -3
- package/dist/fuzz.d.ts.map +1 -1
- package/dist/fuzz.js +4 -10
- package/dist/fuzz.js.map +1 -1
- package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Bn7_xWCv.d.ts} +111 -111
- package/dist/heldout-gate-Bn7_xWCv.d.ts.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/index-Bfs5aufo.d.ts +704 -0
- package/dist/index-Bfs5aufo.d.ts.map +1 -0
- package/dist/{index-CGtH1piv.d.ts → index-CM-SM00y.d.ts} +7 -38
- package/dist/index-CM-SM00y.d.ts.map +1 -0
- package/dist/{index-D-V8gCs_.d.ts → index-DBbivBNs.d.ts} +30 -22
- package/dist/index-DBbivBNs.d.ts.map +1 -0
- package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
- package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
- package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
- package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
- package/dist/index.d.ts +67 -37
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +47 -41
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
- package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
- package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
- package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
- package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
- package/dist/internal-BMFSR8Ns.js.map +1 -0
- package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
- package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
- package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
- package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
- package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
- package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
- package/dist/llm-client-BFMRpmqb.js.map +1 -0
- package/dist/{llm-judge-BhasIPFT.js → llm-judge-DbJdo8Nj.js} +101 -23
- package/dist/llm-judge-DbJdo8Nj.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/{matrix-eXKRMHnL.d.ts → matrix-BpI5Trmo.d.ts} +3 -3
- package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-BpI5Trmo.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +5 -4
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +14 -22
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
- package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +4 -4
- package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
- package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
- package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
- package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
- package/dist/pipelines/index.d.ts +7 -6
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +5 -20
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
- package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
- package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
- package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
- package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
- package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
- package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
- package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
- package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
- package/dist/{produced-state-DZ89riy5.js → produced-state-CtSIp5cQ.js} +5 -5
- package/dist/{produced-state-DZ89riy5.js.map → produced-state-CtSIp5cQ.js.map} +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-CkXSgKkF.d.ts} +3 -3
- package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-CkXSgKkF.d.ts.map} +1 -1
- package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
- package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
- package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
- package/dist/proposal-findings-bko3GGy-.js.map +1 -0
- package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CIRUardl.d.ts} +891 -891
- package/dist/provenance-CIRUardl.d.ts.map +1 -0
- package/dist/{query-CHmMP42p.js → query-BPGMVlbM.js} +46 -4
- package/dist/query-BPGMVlbM.js.map +1 -0
- package/dist/{query-DxPYqpmT.d.ts → query-Na5gEIGd.d.ts} +21 -4
- package/dist/query-Na5gEIGd.d.ts.map +1 -0
- package/dist/random-Dn5fPWkt.js +21 -0
- package/dist/random-Dn5fPWkt.js.map +1 -0
- package/dist/record-id-DUgsK5qp.js +17 -0
- package/dist/record-id-DUgsK5qp.js.map +1 -0
- package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
- package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
- package/dist/{release-confidence-DKfD2RYU.js → release-confidence-CzUHc4z4.js} +4 -4
- package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
- package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
- package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +6 -6
- package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
- package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
- package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
- package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
- package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-SkxYgT0x.js} +3 -3
- package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
- package/dist/rl.d.ts +8 -8
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +13 -18
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
- package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
- package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
- package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
- package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
- package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
- package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
- package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
- package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
- package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
- package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-I36eejJx.js} +3 -3
- package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
- package/dist/{sequential-rYW-Ophm.js → sequential-B51qAYE4.js} +4 -4
- package/dist/{sequential-rYW-Ophm.js.map → sequential-B51qAYE4.js.map} +1 -1
- package/dist/{server-BtFd4uzB.js → server-CCEnywOR.js} +27 -19
- package/dist/server-CCEnywOR.js.map +1 -0
- package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-B2R9C5aG.js} +11 -11
- package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-B2R9C5aG.js.map} +1 -1
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-DFS7QGpS.d.ts} +3 -3
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-DFS7QGpS.d.ts.map} +1 -1
- package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
- package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
- package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
- package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
- package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-B2DJ_82T.d.ts} +102 -36
- package/dist/store-tool-spans-B2DJ_82T.d.ts.map +1 -0
- package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-B9o6tU8f.js} +23 -19
- package/dist/store-tool-spans-B9o6tU8f.js.map +1 -0
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
- package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
- package/dist/{summary-report-BI5hUtvK.js → summary-report-Bgh8CpNK.js} +7 -7
- package/dist/{summary-report-BI5hUtvK.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
- package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
- package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
- package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
- package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-BnXlCJZQ.d.ts} +3 -3
- package/dist/tool-groups-BnXlCJZQ.d.ts.map +1 -0
- package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
- package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
- package/dist/{tool-waste-BqzmVdJk.js → tool-waste-CwGHzBzX.js} +4 -4
- package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +3 -3
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -4
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +60 -62
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +13 -22
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
- package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +4 -14
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-Bfk0uxRj.d.ts.map +1 -1
- package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
- package/dist/types-CiWITkGo.js.map +1 -0
- package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
- package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
- package/dist/{types-D4s7Z6nq.d.ts → types-Dy237wiH.d.ts} +3 -3
- package/dist/{types-D4s7Z6nq.d.ts.map → types-Dy237wiH.d.ts.map} +1 -1
- package/dist/{types-BPb2Kf_C2.d.ts → types-i21ccEkr.d.ts} +3 -3
- package/dist/types-i21ccEkr.d.ts.map +1 -0
- package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
- package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
- package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
- package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
- package/dist/wire/index.d.ts +23 -8
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/code-agent-intake.md +64 -0
- package/docs/concepts.md +1 -1
- package/docs/design/statistics-decisions.md +1 -1
- package/docs/distributed-driver.md +3 -6
- package/docs/public-api.md +122 -106
- package/docs/wire-protocol.md +5 -3
- package/package.json +9 -2
- package/dist/benchmark-BhT16ep9.js.map +0 -1
- package/dist/benchmark-command-BDC3Gocz.js.map +0 -1
- package/dist/campaign-BSmOwskD.js.map +0 -1
- package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
- package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
- package/dist/contract/index.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-DptEII26.js.map +0 -1
- package/dist/emitter-BpYFQPj4.js.map +0 -1
- package/dist/external-optimizer-process-WosTBChy.js.map +0 -1
- package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +0 -1
- package/dist/index-CGtH1piv.d.ts.map +0 -1
- package/dist/index-D-V8gCs_.d.ts.map +0 -1
- package/dist/index-vrJugRal.d.ts +0 -1
- package/dist/internal-BDHPCnjk.js.map +0 -1
- package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
- package/dist/kind-factory-DY8FdoXf.js.map +0 -1
- package/dist/llm-client-hgDieDNN.js.map +0 -1
- package/dist/llm-judge-BhasIPFT.js.map +0 -1
- package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
- package/dist/query-CHmMP42p.js.map +0 -1
- package/dist/query-DxPYqpmT.d.ts.map +0 -1
- package/dist/run-score-lDzV0X8j.js.map +0 -1
- package/dist/server-BtFd4uzB.js.map +0 -1
- package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
- package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
- package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
- package/dist/types-BI4fT3HN.js.map +0 -1
- package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,162 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.170.0] — 2026-08-21
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `pnpm check:collation-ordering`, wired into `verify:package`. It reports a `localeCompare` comparator inside a function that also canonicalizes or hashes — the shape that let eleven orderings decide a digest from the host's collation rather than from the value. The rule is narrow on purpose: ordering a table a human reads is not this gate's business, and a Markdown renderer that sorts rows does not canonicalize, so it is not reported. An allowlist entry matching nothing fails the gate, so a stale waiver cannot hide a new one.
|
|
12
|
+
- `scripts/source-scan.mjs` holds the reading primitives both source gates share — walking `src/`, naming a function by its binding, locating an offset, visiting a tree. A second gate with its own copies would have re-created the duplication these releases removed, and sharing them means the naming fix in 0.168.1 reaches both gates rather than one.
|
|
13
|
+
|
|
14
|
+
---
|
|
15
|
+
|
|
16
|
+
## [0.169.0] — 2026-08-21
|
|
17
|
+
|
|
18
|
+
### Fixed
|
|
19
|
+
|
|
20
|
+
- Eleven orderings that reach a digest, a stable serialization, or a stored identity no longer read the host's collation. RFC 8785 canonicalizes an array **by position**, so a `localeCompare` sort in front of `canonicalString` or `hashCanonical` let the machine decide the digest bytes: the ids `Accuracy, brevity, Clarity` order as `Accuracy,brevity,Clarity` under an en-US collation and as `Accuracy,Clarity,brevity` by code unit, and the two produce different digests for the same data. `compareCodeUnits` in `src/ledger-core/canonical.ts` is now the one comparator for an ordering whose result is hashed.
|
|
21
|
+
- Three of the ten carried a docstring asserting an invariant the code did not hold: `hashScenarios` ("independent of insertion order" — it was not independent of collation), `Dataset.toJsonl` ("deterministic byte-for-byte"), and `serializeFeedbackTrajectoriesJsonl` ("two exports of equal stores are byte-equal"). `testSuiteDigest`'s "two suites with the same digest are the same suite" was in the same position.
|
|
22
|
+
- Digest and serialization sites: `src/dataset.ts` (`hashScenarios`, `toJsonl`), `src/feedback-trajectory.ts` (`serializeFeedbackTrajectoriesJsonl`), `src/trace-repair/test-oracle.ts` (`testSuiteDigest`), `src/campaign/provenance.ts` (`costReceiptsDigest`, `campaignMeasurementDigest` cells), `src/analyst/benchmark-summary.ts` (finding signature), `src/analyst/benchmark-verification-artifacts.ts` (artifact order, which reaches a verification span id), `src/rl/verified-findings-dataset.ts` (dataset rows), `src/campaign/run-campaign.ts` (the campaign's retained cell order).
|
|
23
|
+
- Ordering-only sites fixed for the same reason, since each decides which records a caller sees: `src/campaign/labeled-store/fs-adapter.ts` (which records a `slice(0, count)` returns) and `src/analyst/benchmark-public-data.ts` (which rows a seeded public-benchmark sample selects).
|
|
24
|
+
- No retention window is added, and that is a per-site finding rather than an assumption: nothing in this package stores one of these values and later recomputes it from source data to compare. `verifyLoopProvenanceRecord` re-hashes the record's stored fields, and `assertCampaignSplitIdentity` recomputes a split digest whose builder never sorted. The values move once, like `rubricVersion` in 0.167.0, rather than needing the dual-verify `surfaceHash` required.
|
|
25
|
+
- A comparator for a report a human reads is left alone; `renderSurfaceDiff` and the Markdown renderers keep `localeCompare`.
|
|
26
|
+
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
## [0.168.1] — 2026-08-21
|
|
30
|
+
|
|
31
|
+
### Fixed
|
|
32
|
+
|
|
33
|
+
- `scripts/check-canonical-json.mjs` no longer names a function after the function it is passed to. An arrow in argument position sits behind the text `helper(`, and the name fallback read that as its name; the fabricated name then entered the module's sorter set, so every caller of the REAL function of that name read as sorting. A gate that reports a file and line whose function contains no sort teaches the next reader that it lies, which is how a useful check gets switched off.
|
|
34
|
+
- Names now come from a binding only — a declaration id, a `const`/`let`/`var` binding, or an object-property key — and a function with no binding is reported as `(anonymous)` and located by its line. Only a bound function can be called by name, so only a bound one is registered as a sorting helper.
|
|
35
|
+
|
|
36
|
+
---
|
|
37
|
+
|
|
38
|
+
## [0.168.0] — 2026-08-21
|
|
39
|
+
|
|
40
|
+
### Fixed
|
|
41
|
+
|
|
42
|
+
- A component surface's stored identity no longer depends on the host's collation. `componentSurfaceIdentityMaterial` ordered component names with `localeCompare`, which reads the runtime's default collation rather than the value, and that material feeds `surfaceContentHash` and `surfaceHash` — identities that are **written down** and later recomputed and compared. Three sites do exactly that (`src/campaign/provenance.ts:351`, `src/campaign/presets/run-optimization.ts:658` and `:693`), each of them a fail-closed refusal, so under a different ICU build or locale a legitimate resume or parent selection could be rejected as a surface that does not match its own identity. The material now comes from `canonicalString`, which orders keys by UTF-16 code unit (RFC 8785) — a property of the value alone.
|
|
43
|
+
- `surfaceHashMatches(surface, storedHash)` is the verify path for the retention window. A stored loop key is 16 hex characters with no room for a scheme tag, so unlike an `agent-profile-cell` id it cannot name the scheme that minted it; the matcher tries the current material and then the retired one, which gives the same property — a key minted by an earlier release still matches its own surface, and an edited surface matches neither. The retired builder is private and reachable only from that matcher; nothing mints from it. The three comparison sites now use the matcher instead of `!==`.
|
|
44
|
+
- Every stored component-surface identity moves, not only one whose names sort differently under a collation: RFC 8785 also orders the two top-level keys, so `components` now precedes `schema`. Prompt and code surfaces are unaffected — neither builds material from an ordered key list.
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## [0.167.1] — 2026-08-21
|
|
49
|
+
|
|
50
|
+
### Changed
|
|
51
|
+
|
|
52
|
+
- `waitForActiveHandlers` and `sendJsonIfOpen` had three byte-identical private copies each, one per local HTTP server (`src/analyst/trace-tool-callback.ts`, `src/campaign/external-optimizer-callback.ts`, `src/campaign/external-optimizer-model-proxy.ts`). Both now live in `src/campaign/external-optimizer-http.ts`, the module all three already imported `closeServer`, `listenLocal`, and `sendJson` from. The drain loop re-reads the handler set on every pass because a running handler can register another; a copy that awaited one snapshot would let the caller close the server with work outstanding. No behavior change and no export change: the helpers are internal to that module's consumers.
|
|
53
|
+
|
|
54
|
+
---
|
|
55
|
+
|
|
56
|
+
## [0.167.0] — 2026-08-21
|
|
57
|
+
|
|
58
|
+
### Changed
|
|
59
|
+
|
|
60
|
+
- `rubricVersion` is now `<name>@sha256-rfc8785:<hex>` and comes from `hashCanonical`, this package's only canonical-JSON encoder. It was a 32-bit djb2 hash over a private `stableStringify` — the twelfth canonical-JSON encoder in the tree, and the only one whose key order came from `localeCompare`. `localeCompare` orders by the runtime's collation rather than by the value, so a rubric whose dimension ids differ only in case (`Accuracy`, `brevity`) could serialize in a different order under a different collation and produce a different tag for the same rubric. RFC 8785 orders by UTF-16 code unit, which depends on the value alone.
|
|
61
|
+
- The scheme is named inside the value. A consumer holding a tag from an earlier release reads a different scheme and can tell "this package changed how it hashes" from "the rubric changed"; with a bare hex string the two are indistinguishable. Tags minted before this release (`anti-slop@a4f2b8c1`) are not comparable with these, and nothing in this package verifies a stored tag — `rubricVersion` is compared for equality, never recomputed against a record — so no legacy encoder is retained.
|
|
62
|
+
- `WIRE_VERSION` is `1.1.0`. The request and response shapes are unchanged, so the major stays `1` and every client that checks the major (`clients/python/src/agent_eval_rpc/client.py:139`) keeps working; the minor moves because a response value changed meaning.
|
|
63
|
+
- The Python client carries `rubricVersion` as an opaque `str` and never computes or parses it, so it needs no change beyond the version lockstep.
|
|
64
|
+
- `RUBRIC_VERSION_SCHEME` is exported from `/wire` for a caller that wants to compare a tag's scheme without parsing the string.
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
## [0.166.1] — 2026-08-21
|
|
69
|
+
|
|
70
|
+
### Changed
|
|
71
|
+
|
|
72
|
+
- The seven FNV-1a loops in this package each say what they hash and why they are frozen. They look interchangeable and are not: one hashes IEEE-754 bytes rather than a string (`src/statistics/internal.ts`), one masks each code unit to a byte and the rest do not (`src/partition-held-out.ts`), one is deliberately left signed so subtracting two results is a valid comparator (`src/analyst/benchmark.ts`), and one runs two accumulators with different offsets to make a 64-bit-wide id (`src/eval-campaign.ts`). Unifying them would move held-out splits, cell keys, replay seeds, and published bootstrap intervals that were already decided under the current values, so each now carries the one line that stops the next reader from "fixing" it. No behavior change.
|
|
73
|
+
|
|
74
|
+
---
|
|
75
|
+
|
|
76
|
+
## [0.165.0] — 2026-08-21
|
|
77
|
+
|
|
78
|
+
### Added
|
|
79
|
+
|
|
80
|
+
- `@tangle-network/agent-eval/adapters/http` is published. `httpDispatch` (coordinator side) and `runDispatchServer` (worker side) have existed in `src/adapters/http.ts` with a test, a two-process runnable example under `examples/distributed-driver/`, and a design document since 0.130.x, but no `exports` entry — so every `import` line in `docs/distributed-driver.md` threw `ERR_PACKAGE_PATH_NOT_EXPORTED` against an installed copy, and the example could only be run from a checkout. The document's status note and the three "not currently published" fence comments are removed because they are no longer true.
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## [0.164.2] — 2026-08-21
|
|
85
|
+
|
|
86
|
+
### Changed
|
|
87
|
+
|
|
88
|
+
- Three vocabularies that were declared twice now derive from one array each. No member changes, no reordering, and no public export changes; the schemas that validate them are generated from the same declaration they validate against, so the two halves cannot drift apart.
|
|
89
|
+
- `AnalystComparisonMetric` was a hand-written 18-member union, `METRICS` repeated the same 18 members as an array, `LOWER_IS_BETTER` repeated 8 of them as a set, and `src/analyst/benchmark-command-validation.ts` repeated all 18 again as a `z.enum` over the persisted artifact. A metric added to the union but missed in that enum would have been rejected at artifact validation with no compile error. One `ANALYST_COMPARISON_METRIC_DIRECTION` table now carries every metric and its direction; the type, the reporting order, the artifact schema, and `analystComparisonMetricDirection` all derive from it.
|
|
90
|
+
- `FailureClass` was a 35-member union followed by `FAILURE_CLASSES`, a `readonly FailureClass[]` repeating the same 35 members. The array could silently omit a member the type declared. `FAILURE_CLASSES` is now the `as const` owner and the type derives from it.
|
|
91
|
+
- `AnalystSeverity` was a union in `src/analyst/types.ts` and `ANALYST_SEVERITIES` was a private copy in `src/analyst/finding-signature.ts` feeding the finding schema's `z.enum`. `ANALYST_SEVERITIES` now lives beside the type it defines.
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## [0.164.1] — 2026-08-21
|
|
96
|
+
|
|
97
|
+
### Changed
|
|
98
|
+
|
|
99
|
+
- The analyst-benchmark pin is authored by the tool that checks it. `pnpm analyst:pin` (`--write`) rewrites `ANALYST_BENCHMARK_IMPLEMENTATION_FILES` from the import graph and both live digests from the sources; it refuses a `--source-root` and never writes `ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256` or `ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256`, which state facts about already-published evidence. Every check the gate made still runs unchanged in every other mode.
|
|
100
|
+
- `src/analyst/benchmark-implementation.test.ts` no longer keeps a second hand-copied copy of the 108-entry source manifest. `assertCompleteSourceManifest` in the checker already proves the manifest equals the import graph on every `pnpm build` and `pnpm verify:package`; the test's `EXPECTED_FILES` restated the constant it imported, so adding one source file meant editing the same list in two places. The two tests that prove the gate actually fails — a mutated bound source, and a transitive source missing from the manifest — are unchanged and now read the constant.
|
|
101
|
+
|
|
102
|
+
---
|
|
103
|
+
|
|
104
|
+
## [0.164.0] — 2026-08-21
|
|
105
|
+
|
|
106
|
+
### Removed
|
|
107
|
+
|
|
108
|
+
- `createVerifierAdapter`, `createRunCriticAdapter`, and `createJudgeAdapter` from `/analyst`, with their `VerifierAdapterOpts`, `RunCriticAdapterOpts`, and `JudgeAdapterOpts`. Every channel was silent: no import or mention across the 31 repositories the census sweeps on their default branches, no hit in a GitHub code search across the organisation outside this repository's own source and changelog, nothing in the published `dist/` of `@tangle-network/agent-runtime@0.153.1`, `@tangle-network/agent-knowledge@10.6.0`, or `@tangle-network/braid@0.2.0`, and no test, example, doc, or in-repository caller. `createSemanticConceptJudgeAdapter` is the remaining adapter and is unchanged.
|
|
109
|
+
|
|
110
|
+
### Added
|
|
111
|
+
|
|
112
|
+
- `docs/code-agent-intake.md`. The per-harness intake functions — `fromCodexSession`, `fromClaudeCodeSession`, `fromOpenCodeSession`, `fromKimiCodeSession`, `fromPiSession` — plus `parseCodeAgentJsonl`, `parseCodeAgentJsonlFile`, `streamCodeAgentJsonlFile`, `observeCodeAgentSession`, `parseAgentTrace`, and `partitionRunsByAuthoringModel` had no Markdown front door, which is the only reason the census read them as unused. The document states the three steps, names the fields on `CodeAgentSessionDiagnostic` and `CodeAgentSessionMetrics` that a caller must read before aggregating, and says which path to use above V8's string ceiling. Linked from the README documentation index.
|
|
113
|
+
|
|
114
|
+
### Fixed
|
|
115
|
+
|
|
116
|
+
- `pnpm api:census` counted `scripts/` and `benchmarks/` as no evidence at all. Neither directory was walked, so a symbol whose only caller is a maintenance script read as a deletion candidate — including `renderEvidenceIndex`, which `pnpm evidence:check` calls inside `pnpm verify:package`. Acting on that row would have failed the release gate. Both directories are now walked and count as in-package production callers, and `SOURCE_FILE` accepts `.mjs`/`.cjs`/`.js` so a `.mjs` script is read at all. Eleven symbols moved from `none` to `production` on this pass.
|
|
117
|
+
- `docs/public-api.md` now records the four rules that decided the `none` review, so the next reader inherits the judgment rather than repeating the sweep. Totals moved from 903/211/238 (production/planned/none) to 926/205/220.
|
|
118
|
+
|
|
119
|
+
---
|
|
120
|
+
|
|
121
|
+
## [0.163.2] — 2026-08-21
|
|
122
|
+
|
|
123
|
+
### Fixed
|
|
124
|
+
|
|
125
|
+
- `scoreKnowledgeReadiness` and `acquisitionPlansForKnowledgeGaps` no longer emit an optional key with no value. `KnowledgeBundle.metadata` and `DataAcquisitionPlan.questions` are both declared optional, but the bundle set `metadata: options.metadata` and every non-`ask_user` plan set `questions: undefined`, so a caller that omitted them got a key present and undefined rather than an absent key. `runAgentControlLoop` fingerprints loop state with RFC 8785 canonical JSON, which refuses an `undefined`-valued field so that two states differing only in such a field cannot hash alike — so a readiness report placed in loop state aborted the run with `value has no canonical JSON form: $.metadata is undefined` and no step was ever taken. Measured downstream: every Agent Knowledge control-loop run through `createKnowledgeControlLoopAdapter` failed this way. Both sites now use the omit form already used by `reference-replay.ts` and the two measured-comparison builders.
|
|
126
|
+
|
|
127
|
+
### Changed
|
|
128
|
+
|
|
129
|
+
- `ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256` is `a9eb17be…a7ef623d` for this release's manifests. The historical `ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256` is unchanged.
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
## [0.163.1] — 2026-08-21
|
|
134
|
+
|
|
135
|
+
### Changed
|
|
136
|
+
|
|
137
|
+
- Every byte-identical private copy of a shared primitive now routes through one owner. No public export changes and no behavior changes.
|
|
138
|
+
- `mulberry32` had five implementations: the exported owner in `src/statistics/random.ts`, named private copies in `src/trajectory-replay/batch.ts` and `src/judge-calibration.ts`, and unnamed inline copies in `src/rl/contamination.ts` (`shuffleOrder`) and `src/fuzz/explorer.ts` (`FuzzExplorer.rng`). The copies used `seed >>> 0` where the owner uses `seed | 0` — the same 32 bits, so the streams are identical (verified over 150,000 draws across 15 seeds). All four are deleted. The replay fix-case sample, the judge-calibration bootstrap, the contamination probe's order-shuffle perturbation, and the fuzz explorer now inherit the owner's refusal of a non-finite seed instead of silently seeding from `NaN >>> 0 === 0`. `src/statistics/random.test.ts` pins the stream with known-answer vectors, so a future edit to the PRNG cannot silently move an already-published statistic.
|
|
139
|
+
- `toAttributes` and `msToNs` were byte-identical in `src/trace/otel.ts` and `src/trace/otel-export.ts`. Both exporters emit the same `OtlpSpan`, so two copies of the value-tagging rule meant the same attribute could be typed differently depending on which exporter ran. They are now `toOtlpAttributes` and `msToUnixNano` in the private `src/trace/otlp-encoding.ts`.
|
|
140
|
+
- `cryptoRandomId` / `cryptoEventId` / `cryptoId` were three byte-identical id mints in `src/trace/emitter.ts`, `src/llm-client.ts`, and `src/builder-eval/builder-session.ts`. Each carried a `Date.now()`-plus-`Math.random()` fallback that is unreachable on every runtime this package supports (`engines.node >= 20`, where `globalThis.crypto.randomUUID` is always present). One private `newRecordId` in `src/record-id.ts` replaces all three, and the weaker id path is gone.
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
---
|
|
146
|
+
|
|
147
|
+
## [0.163.0] — 2026-08-21
|
|
148
|
+
|
|
149
|
+
### Fixed
|
|
150
|
+
|
|
151
|
+
- `calibrationCurve` measured the metric the caller named instead of always reading the run score. Its private extractor returned `run.outcome.score` whenever a score was present, for every metric id, and the report was then stamped with the requested `evalMetric` — so a curve labelled `costUsd` or `durationMs` plotted the score. `correlationStudy`'s separate copy returned `null` for `outputTokens` and `failureClass`, which dropped the pair from the study rather than reporting it, and an unrecognized metric id produced an empty study in both entry points instead of a refusal. Three private copies of one extractor had drifted; `runMetricExtractor` in `src/trace/query.ts` is now the only one, and an id outside `RUN_METRICS` raises `ValidationError` naming the built-in metrics.
|
|
152
|
+
|
|
153
|
+
### Added
|
|
154
|
+
|
|
155
|
+
- `RUN_METRICS`, `RunMetric`, `isRunMetric`, and `runMetricExtractor` on `/traces`. `RUN_METRICS` is the vocabulary `regressionView`, `correlationStudy`, and `calibrationCurve` read without a caller-supplied `extract`: `score`, `overallScore`, `pass`, `durationMs`, `costUsd`, `inputTokens`, `outputTokens`, `failureClass`. `RunMetric` derives from the array, so a metric cannot be declared without an extractor arm.
|
|
156
|
+
|
|
157
|
+
### Changed
|
|
158
|
+
|
|
159
|
+
- `ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256` is `114f0786…eeb69400` for this release's manifests. The historical `ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256` is unchanged.
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
7
163
|
## [0.161.1] — 2026-08-21
|
|
8
164
|
|
|
9
165
|
### Changed
|
package/README.md
CHANGED
|
@@ -153,6 +153,7 @@ and [DSPy](./docs/campaign-proposers.md#use-official-dspy-optimizers).
|
|
|
153
153
|
| `@tangle-network/agent-eval/benchmarks` | Benchmark adapters and retrieval metrics. |
|
|
154
154
|
| `@tangle-network/agent-eval/rl` | Export rewards, preferences, and training rows. |
|
|
155
155
|
| `@tangle-network/agent-eval/wire` | HTTP and RPC schemas for other languages. |
|
|
156
|
+
| `@tangle-network/agent-eval/adapters/http` | Run campaign cells on remote workers over HTTP. |
|
|
156
157
|
|
|
157
158
|
Use the root import for common primitives.
|
|
158
159
|
Use a subpath when you want an explicit capability boundary.
|
|
@@ -169,6 +170,7 @@ Use a subpath when you want an explicit capability boundary.
|
|
|
169
170
|
| How do I register an experiment as a sealed object? | [`docs/experiment.md`](./docs/experiment.md) |
|
|
170
171
|
| How is something certified without an answer key? | [`docs/verification-strategies.md`](./docs/verification-strategies.md) |
|
|
171
172
|
| Where does every verifier land its result? | [`docs/verdicts.md`](./docs/verdicts.md) |
|
|
173
|
+
| How do I turn a coding-agent session log into runs? | [`docs/code-agent-intake.md`](./docs/code-agent-intake.md) |
|
|
172
174
|
| How do I score a string from another language? | [`docs/wire-protocol.md`](./docs/wire-protocol.md) |
|
|
173
175
|
|
|
174
176
|
The [example index](./examples/README.md) lists every runnable example.
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { i as makeRng } from "./internal-
|
|
1
|
+
import { i as makeRng } from "./internal-BMFSR8Ns.js";
|
|
2
2
|
import { n as observedScore } from "./reward-nw2xZGZG.js";
|
|
3
3
|
//#region src/rl/active-curriculum.ts
|
|
4
4
|
/**
|
|
@@ -201,4 +201,4 @@ function sampleGamma(shape, rng) {
|
|
|
201
201
|
//#endregion
|
|
202
202
|
export { thompsonCurriculum as n, varianceBasedCurriculum as r, observationsFromRunRecords as t };
|
|
203
203
|
|
|
204
|
-
//# sourceMappingURL=active-curriculum-
|
|
204
|
+
//# sourceMappingURL=active-curriculum-OjIWgrUJ.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"active-curriculum-CD5TU2yW.js","names":[],"sources":["../src/rl/active-curriculum.ts"],"sourcesContent":["/**\n * Adaptive curriculum / active scenario selection.\n *\n * Fixed scenario sets waste sample budget on cells the policy already\n * passes (no information left) and cells the policy never passes (no\n * gradient available either). Active learning over scenarios fixes this\n * by allocating the next sample budget to cells where the policy's\n * outcome is *uncertain* — those carry the most decision-relevant signal.\n *\n * This module ships two complementary strategies:\n *\n * 1. **Variance-based** — score each (variant, scenario) cell by the\n * empirical variance of past observations. Allocate next-round budget\n * proportional to variance. Standard active-learning-by-uncertainty\n * heuristic; works well when the policy is non-deterministic and\n * cells differ in observation noise.\n *\n * 2. **Bandit-based (Thompson sampling)** — model each (variant,\n * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick\n * cells whose posterior mean is closest to the per-scenario decision\n * threshold. The right primitive when scenarios are\n * \"pass/fail\" rather than continuous, and when promotion gates fire\n * at a known threshold (e.g., 0.5).\n *\n * The output is a *next-round budget allocation* — a list of (variant,\n * scenario, count) triples. The consumer's matrix runner consumes the\n * allocation, runs those cells, feeds the new observations back. Loop.\n *\n * Out of scope (deliberate): scenario *generation* — that's the\n * adversarial primitive's job. This module allocates over an existing\n * scenario pool.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { makeRng } from '../statistics/internal'\n\nexport interface CellObservation {\n variantId: string\n scenarioId: string\n /** Observed score in [0, 1]. */\n score: number\n /** For Bernoulli arms — derive from the score with a threshold if needed. */\n pass?: boolean\n}\n\nexport interface CurriculumAllocation {\n variantId: string\n scenarioId: string\n /** How many additional reps to run on this cell. */\n count: number\n /** Strategy-specific reason for the allocation. */\n reason: string\n}\n\nexport interface VarianceCurriculumOptions {\n /** Total reps to allocate across all cells. */\n budget: number\n /**\n * Smoothing prior on variance — keeps the allocator from concentrating\n * on a cell with one observation just because its 1-sample variance is\n * 0. Default 0.05.\n */\n variancePrior?: number\n /**\n * Minimum reps per cell — even when the variance estimate is low, give\n * every cell at least this many. Default 1.\n */\n floorPerCell?: number\n}\n\n/**\n * Variance-proportional allocation. For each cell, estimate variance from\n * past observations + a prior, then allocate the budget proportional to\n * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule\n * (Neyman 1934) that balances \"explore noisy cells\" with \"explore\n * under-sampled cells.\"\n */\nexport function varianceBasedCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: VarianceCurriculumOptions,\n): CurriculumAllocation[] {\n const variancePrior = opts.variancePrior ?? 0.05\n const floor = opts.floorPerCell ?? 1\n const budget = opts.budget\n\n const grouped = new Map<string, number[]>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const arr = grouped.get(k) ?? []\n arr.push(o.score)\n grouped.set(k, arr)\n }\n\n const cellStats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const samples = grouped.get(k) ?? []\n const n = samples.length\n const mean = n === 0 ? 0.5 : samples.reduce((s, v) => s + v, 0) / n\n const variance =\n n < 2\n ? variancePrior\n : samples.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1) + variancePrior\n // Neyman optimal allocation: weight ∝ √variance; add √(1/n) to break\n // ties toward under-sampled cells.\n const weight = Math.sqrt(variance) + 1 / Math.sqrt(Math.max(1, n))\n return { variantId: c.variantId, scenarioId: c.scenarioId, n, mean, variance, weight }\n })\n\n // Reserve floor*N for the floor; allocate the rest proportional to weight.\n const floorTotal = floor * cellStats.length\n if (floorTotal >= budget) {\n const each = Math.max(1, Math.floor(budget / Math.max(1, cellStats.length)))\n return cellStats.map((c) => ({\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: each,\n reason: `floor allocation (budget tight; n=${c.n})`,\n }))\n }\n const remaining = budget - floorTotal\n const totalWeight = cellStats.reduce((s, c) => s + c.weight, 0)\n return cellStats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * remaining)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: floor + proportional,\n reason: `variance ${c.variance.toFixed(3)} (n=${c.n}, mean=${c.mean.toFixed(3)})`,\n }\n })\n}\n\nexport interface ThompsonCurriculumOptions {\n budget: number\n /**\n * The per-scenario decision threshold. Cells whose posterior mean is\n * closest to this get the most budget — that's where the next observation\n * has the highest information value for the gate decision. Default 0.5.\n */\n decisionThreshold?: number\n /** Beta prior parameters. Default α=β=1 (uniform). */\n priorAlpha?: number\n priorBeta?: number\n /** Seed the Thompson sampler. Absent, the seed is derived from the observed\n * scores, so the same observations reproduce the same allocation. */\n seed?: number\n}\n\n/**\n * Thompson-sampling-style allocation for pass/fail cells. For each cell:\n *\n * - Maintain Beta(α + passes, β + failures) posterior on pass-rate\n * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):\n * cells whose sampled posterior straddles the decision boundary get\n * the most weight; cells already clearly above or below get less.\n *\n * This is the right primitive when promotion gates fire at a known\n * threshold and you want to sharpen the posterior near the boundary.\n */\nexport function thompsonCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: ThompsonCurriculumOptions,\n): CurriculumAllocation[] {\n const threshold = opts.decisionThreshold ?? 0.5\n const alpha0 = opts.priorAlpha ?? 1\n const beta0 = opts.priorBeta ?? 1\n const rng = makeRng(\n opts.seed,\n observations.map((observation) => observation.score),\n )\n\n const grouped = new Map<string, { passes: number; failures: number }>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const pass = o.pass ?? o.score >= threshold\n if (pass) cur.passes += 1\n else cur.failures += 1\n grouped.set(k, cur)\n }\n\n const stats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const a = alpha0 + cur.passes\n const b = beta0 + cur.failures\n // Sample a single Beta draw — the Thompson signal.\n const sampled = sampleBeta(a, b, rng)\n const distance = Math.abs(sampled - threshold)\n // Information-near-threshold weight: closer = higher.\n // Use Gaussian-shaped kernel with σ tuned to posterior std.\n const variance = (a * b) / ((a + b) ** 2 * (a + b + 1))\n const sigma = Math.max(0.05, Math.sqrt(variance))\n const weight = Math.exp(-((distance / sigma) ** 2))\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n n: cur.passes + cur.failures,\n sampled,\n sigma,\n weight,\n a,\n b,\n }\n })\n\n const totalWeight = stats.reduce((s, c) => s + c.weight, 0)\n return stats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * opts.budget)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: Math.max(0, proportional),\n reason: `Beta(${c.a.toFixed(1)},${c.b.toFixed(1)}) sample=${c.sampled.toFixed(3)} (target ${threshold})`,\n }\n })\n}\n\n/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */\nexport function observationsFromRunRecords(\n runs: RunRecord[],\n opts: { passThreshold?: number; useHoldout?: boolean } = {},\n): CellObservation[] {\n const threshold = opts.passThreshold ?? 0.5\n const useHoldout = opts.useHoldout ?? true\n const out: CellObservation[] = []\n for (const r of runs) {\n if (!r.scenarioId) continue\n // Ungated on purpose, and the precedence is caller policy, not a default:\n // `useHoldout: false` means \"score this curriculum on the search split when\n // both exist\". This feeds sampling COUNTS, not an exported reward. Known\n // risk: a gamed run's high score inflates the cell's Beta posterior, so the\n // curriculum stops sampling a cell it wrongly believes is solved. The fix\n // for that is an upstream filter on gated records — zeroing the score here\n // would push the posterior the opposite way and be equally wrong.\n const score = observedScore(r, useHoldout ? 'holdout' : 'search')\n if (typeof score !== 'number' || !Number.isFinite(score)) continue\n out.push({\n variantId: r.candidateId,\n scenarioId: r.scenarioId,\n score,\n pass: score >= threshold,\n })\n }\n return out\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\n/**\n * Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma\n * variates. Accuracy is good for α, β > 1; we floor the parameters at 1\n * to avoid degenerate cases.\n */\nfunction sampleBeta(alpha: number, beta: number, rng: () => number): number {\n const a = Math.max(1, alpha)\n const b = Math.max(1, beta)\n const x = sampleGamma(a, rng)\n const y = sampleGamma(b, rng)\n return x / (x + y)\n}\n\nfunction sampleGamma(shape: number, rng: () => number): number {\n // Marsaglia–Tsang for shape ≥ 1.\n const d = shape - 1 / 3\n const c = 1 / Math.sqrt(9 * d)\n while (true) {\n let x: number\n let v: number\n do {\n const u1 = rng() || 1e-12\n const u2 = rng() || 1e-12\n // Box-Muller for a normal sample.\n x = Math.sqrt(-2 * Math.log(u1)) * Math.cos(2 * Math.PI * u2)\n v = 1 + c * x\n } while (v <= 0)\n v = v * v * v\n const u = rng()\n if (u < 1 - 0.0331 * x ** 4) return d * v\n if (Math.log(u) < 0.5 * x * x + d * (1 - v + Math.log(v))) return d * v\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8EA,SAAgB,wBACd,cACA,gBACA,MACwB;CACxB,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,QAAQ,KAAK,gBAAgB;CACnC,MAAM,SAAS,KAAK;CAEpB,MAAM,0BAAU,IAAI,IAAsB;CAC1C,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,CAAC;EAC/B,IAAI,KAAK,EAAE,KAAK;EAChB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,YAAY,eAAe,KAAK,MAAM;EAC1C,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,UAAU,QAAQ,IAAI,CAAC,KAAK,CAAC;EACnC,MAAM,IAAI,QAAQ;EAClB,MAAM,OAAO,MAAM,IAAI,KAAM,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAClE,MAAM,WACJ,IAAI,IACA,gBACA,QAAQ,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI,KAAK;EAGnE,MAAM,SAAS,KAAK,KAAK,QAAQ,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,CAAC,CAAC;EACjE,OAAO;GAAE,WAAW,EAAE;GAAW,YAAY,EAAE;GAAY;GAAG;GAAM;GAAU;EAAO;CACvF,CAAC;CAGD,MAAM,aAAa,QAAQ,UAAU;CACrC,IAAI,cAAc,QAAQ;EACxB,MAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,SAAS,KAAK,IAAI,GAAG,UAAU,MAAM,CAAC,CAAC;EAC3E,OAAO,UAAU,KAAK,OAAO;GAC3B,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO;GACP,QAAQ,qCAAqC,EAAE,EAAE;EACnD,EAAE;CACJ;CACA,MAAM,YAAY,SAAS;CAC3B,MAAM,cAAc,UAAU,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC9D,OAAO,UAAU,KAAK,MAAM;EAC1B,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,SAAS;EAC5F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,QAAQ;GACf,QAAQ,YAAY,EAAE,SAAS,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE,SAAS,EAAE,KAAK,QAAQ,CAAC,EAAE;EACjF;CACF,CAAC;AACH;;;;;;;;;;;;AA6BA,SAAgB,mBACd,cACA,gBACA,MACwB;CACxB,MAAM,YAAY,KAAK,qBAAqB;CAC5C,MAAM,SAAS,KAAK,cAAc;CAClC,MAAM,QAAQ,KAAK,aAAa;CAChC,MAAM,MAAM,QACV,KAAK,MACL,aAAa,KAAK,gBAAgB,YAAY,KAAK,CACrD;CAEA,MAAM,0BAAU,IAAI,IAAkD;CACtE,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EAEvD,IADa,EAAE,QAAQ,EAAE,SAAS,WACxB,IAAI,UAAU;OACnB,IAAI,YAAY;EACrB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,QAAQ,eAAe,KAAK,MAAM;EACtC,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EACvD,MAAM,IAAI,SAAS,IAAI;EACvB,MAAM,IAAI,QAAQ,IAAI;EAEtB,MAAM,UAAU,WAAW,GAAG,GAAG,GAAG;EACpC,MAAM,WAAW,KAAK,IAAI,UAAU,SAAS;EAG7C,MAAM,WAAY,IAAI,MAAO,IAAI,MAAM,KAAK,IAAI,IAAI;EACpD,MAAM,QAAQ,KAAK,IAAI,KAAM,KAAK,KAAK,QAAQ,CAAC;EAChD,MAAM,SAAS,KAAK,IAAI,GAAG,WAAW,UAAU,EAAE;EAClD,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,GAAG,IAAI,SAAS,IAAI;GACpB;GACA;GACA;GACA;GACA;EACF;CACF,CAAC;CAED,MAAM,cAAc,MAAM,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC1D,OAAO,MAAM,KAAK,MAAM;EACtB,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,KAAK,MAAM;EAC9F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,KAAK,IAAI,GAAG,YAAY;GAC/B,QAAQ,QAAQ,EAAE,EAAE,QAAQ,CAAC,EAAE,GAAG,EAAE,EAAE,QAAQ,CAAC,EAAE,WAAW,EAAE,QAAQ,QAAQ,CAAC,EAAE,WAAW,UAAU;EACxG;CACF,CAAC;AACH;;AAGA,SAAgB,2BACd,MACA,OAAyD,CAAC,GACvC;CACnB,MAAM,YAAY,KAAK,iBAAiB;CACxC,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,MAAyB,CAAC;CAChC,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,CAAC,EAAE,YAAY;EAQnB,MAAM,QAAQ,cAAc,GAAG,aAAa,YAAY,QAAQ;EAChE,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EAC1D,IAAI,KAAK;GACP,WAAW,EAAE;GACb,YAAY,EAAE;GACd;GACA,MAAM,SAAS;EACjB,CAAC;CACH;CACA,OAAO;AACT;;;;;;AASA,SAAS,WAAW,OAAe,MAAc,KAA2B;CAC1E,MAAM,IAAI,KAAK,IAAI,GAAG,KAAK;CAC3B,MAAM,IAAI,KAAK,IAAI,GAAG,IAAI;CAC1B,MAAM,IAAI,YAAY,GAAG,GAAG;CAE5B,OAAO,KAAK,IADF,YAAY,GAAG,GACT;AAClB;AAEA,SAAS,YAAY,OAAe,KAA2B;CAE7D,MAAM,IAAI,QAAQ,IAAI;CACtB,MAAM,IAAI,IAAI,KAAK,KAAK,IAAI,CAAC;CAC7B,OAAO,MAAM;EACX,IAAI;EACJ,IAAI;EACJ,GAAG;GACD,MAAM,KAAK,IAAI,KAAK;GACpB,MAAM,KAAK,IAAI,KAAK;GAEpB,IAAI,KAAK,KAAK,KAAK,KAAK,IAAI,EAAE,CAAC,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;GAC5D,IAAI,IAAI,IAAI;EACd,SAAS,KAAK;EACd,IAAI,IAAI,IAAI;EACZ,MAAM,IAAI,IAAI;EACd,IAAI,IAAI,IAAI,QAAS,KAAK,GAAG,OAAO,IAAI;EACxC,IAAI,KAAK,IAAI,CAAC,IAAI,KAAM,IAAI,IAAI,KAAK,IAAI,IAAI,KAAK,IAAI,CAAC,IAAI,OAAO,IAAI;CACxE;AACF"}
|
|
1
|
+
{"version":3,"file":"active-curriculum-OjIWgrUJ.js","names":[],"sources":["../src/rl/active-curriculum.ts"],"sourcesContent":["/**\n * Adaptive curriculum / active scenario selection.\n *\n * Fixed scenario sets waste sample budget on cells the policy already\n * passes (no information left) and cells the policy never passes (no\n * gradient available either). Active learning over scenarios fixes this\n * by allocating the next sample budget to cells where the policy's\n * outcome is *uncertain* — those carry the most decision-relevant signal.\n *\n * This module ships two complementary strategies:\n *\n * 1. **Variance-based** — score each (variant, scenario) cell by the\n * empirical variance of past observations. Allocate next-round budget\n * proportional to variance. Standard active-learning-by-uncertainty\n * heuristic; works well when the policy is non-deterministic and\n * cells differ in observation noise.\n *\n * 2. **Bandit-based (Thompson sampling)** — model each (variant,\n * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick\n * cells whose posterior mean is closest to the per-scenario decision\n * threshold. The right primitive when scenarios are\n * \"pass/fail\" rather than continuous, and when promotion gates fire\n * at a known threshold (e.g., 0.5).\n *\n * The output is a *next-round budget allocation* — a list of (variant,\n * scenario, count) triples. The consumer's matrix runner consumes the\n * allocation, runs those cells, feeds the new observations back. Loop.\n *\n * Out of scope (deliberate): scenario *generation* — that's the\n * adversarial primitive's job. This module allocates over an existing\n * scenario pool.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { makeRng } from '../statistics/internal'\n\nexport interface CellObservation {\n variantId: string\n scenarioId: string\n /** Observed score in [0, 1]. */\n score: number\n /** For Bernoulli arms — derive from the score with a threshold if needed. */\n pass?: boolean\n}\n\nexport interface CurriculumAllocation {\n variantId: string\n scenarioId: string\n /** How many additional reps to run on this cell. */\n count: number\n /** Strategy-specific reason for the allocation. */\n reason: string\n}\n\nexport interface VarianceCurriculumOptions {\n /** Total reps to allocate across all cells. */\n budget: number\n /**\n * Smoothing prior on variance — keeps the allocator from concentrating\n * on a cell with one observation just because its 1-sample variance is\n * 0. Default 0.05.\n */\n variancePrior?: number\n /**\n * Minimum reps per cell — even when the variance estimate is low, give\n * every cell at least this many. Default 1.\n */\n floorPerCell?: number\n}\n\n/**\n * Variance-proportional allocation. For each cell, estimate variance from\n * past observations + a prior, then allocate the budget proportional to\n * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule\n * (Neyman 1934) that balances \"explore noisy cells\" with \"explore\n * under-sampled cells.\"\n */\nexport function varianceBasedCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: VarianceCurriculumOptions,\n): CurriculumAllocation[] {\n const variancePrior = opts.variancePrior ?? 0.05\n const floor = opts.floorPerCell ?? 1\n const budget = opts.budget\n\n const grouped = new Map<string, number[]>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const arr = grouped.get(k) ?? []\n arr.push(o.score)\n grouped.set(k, arr)\n }\n\n const cellStats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const samples = grouped.get(k) ?? []\n const n = samples.length\n const mean = n === 0 ? 0.5 : samples.reduce((s, v) => s + v, 0) / n\n const variance =\n n < 2\n ? variancePrior\n : samples.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1) + variancePrior\n // Neyman optimal allocation: weight ∝ √variance; add √(1/n) to break\n // ties toward under-sampled cells.\n const weight = Math.sqrt(variance) + 1 / Math.sqrt(Math.max(1, n))\n return { variantId: c.variantId, scenarioId: c.scenarioId, n, mean, variance, weight }\n })\n\n // Reserve floor*N for the floor; allocate the rest proportional to weight.\n const floorTotal = floor * cellStats.length\n if (floorTotal >= budget) {\n const each = Math.max(1, Math.floor(budget / Math.max(1, cellStats.length)))\n return cellStats.map((c) => ({\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: each,\n reason: `floor allocation (budget tight; n=${c.n})`,\n }))\n }\n const remaining = budget - floorTotal\n const totalWeight = cellStats.reduce((s, c) => s + c.weight, 0)\n return cellStats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * remaining)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: floor + proportional,\n reason: `variance ${c.variance.toFixed(3)} (n=${c.n}, mean=${c.mean.toFixed(3)})`,\n }\n })\n}\n\nexport interface ThompsonCurriculumOptions {\n budget: number\n /**\n * The per-scenario decision threshold. Cells whose posterior mean is\n * closest to this get the most budget — that's where the next observation\n * has the highest information value for the gate decision. Default 0.5.\n */\n decisionThreshold?: number\n /** Beta prior parameters. Default α=β=1 (uniform). */\n priorAlpha?: number\n priorBeta?: number\n /** Seed the Thompson sampler. Absent, the seed is derived from the observed\n * scores, so the same observations reproduce the same allocation. */\n seed?: number\n}\n\n/**\n * Thompson-sampling-style allocation for pass/fail cells. For each cell:\n *\n * - Maintain Beta(α + passes, β + failures) posterior on pass-rate\n * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):\n * cells whose sampled posterior straddles the decision boundary get\n * the most weight; cells already clearly above or below get less.\n *\n * This is the right primitive when promotion gates fire at a known\n * threshold and you want to sharpen the posterior near the boundary.\n */\nexport function thompsonCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: ThompsonCurriculumOptions,\n): CurriculumAllocation[] {\n const threshold = opts.decisionThreshold ?? 0.5\n const alpha0 = opts.priorAlpha ?? 1\n const beta0 = opts.priorBeta ?? 1\n const rng = makeRng(\n opts.seed,\n observations.map((observation) => observation.score),\n )\n\n const grouped = new Map<string, { passes: number; failures: number }>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const pass = o.pass ?? o.score >= threshold\n if (pass) cur.passes += 1\n else cur.failures += 1\n grouped.set(k, cur)\n }\n\n const stats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const a = alpha0 + cur.passes\n const b = beta0 + cur.failures\n // Sample a single Beta draw — the Thompson signal.\n const sampled = sampleBeta(a, b, rng)\n const distance = Math.abs(sampled - threshold)\n // Information-near-threshold weight: closer = higher.\n // Use Gaussian-shaped kernel with σ tuned to posterior std.\n const variance = (a * b) / ((a + b) ** 2 * (a + b + 1))\n const sigma = Math.max(0.05, Math.sqrt(variance))\n const weight = Math.exp(-((distance / sigma) ** 2))\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n n: cur.passes + cur.failures,\n sampled,\n sigma,\n weight,\n a,\n b,\n }\n })\n\n const totalWeight = stats.reduce((s, c) => s + c.weight, 0)\n return stats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * opts.budget)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: Math.max(0, proportional),\n reason: `Beta(${c.a.toFixed(1)},${c.b.toFixed(1)}) sample=${c.sampled.toFixed(3)} (target ${threshold})`,\n }\n })\n}\n\n/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */\nexport function observationsFromRunRecords(\n runs: RunRecord[],\n opts: { passThreshold?: number; useHoldout?: boolean } = {},\n): CellObservation[] {\n const threshold = opts.passThreshold ?? 0.5\n const useHoldout = opts.useHoldout ?? true\n const out: CellObservation[] = []\n for (const r of runs) {\n if (!r.scenarioId) continue\n // Ungated on purpose, and the precedence is caller policy, not a default:\n // `useHoldout: false` means \"score this curriculum on the search split when\n // both exist\". This feeds sampling COUNTS, not an exported reward. Known\n // risk: a gamed run's high score inflates the cell's Beta posterior, so the\n // curriculum stops sampling a cell it wrongly believes is solved. The fix\n // for that is an upstream filter on gated records — zeroing the score here\n // would push the posterior the opposite way and be equally wrong.\n const score = observedScore(r, useHoldout ? 'holdout' : 'search')\n if (typeof score !== 'number' || !Number.isFinite(score)) continue\n out.push({\n variantId: r.candidateId,\n scenarioId: r.scenarioId,\n score,\n pass: score >= threshold,\n })\n }\n return out\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\n/**\n * Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma\n * variates. Accuracy is good for α, β > 1; we floor the parameters at 1\n * to avoid degenerate cases.\n */\nfunction sampleBeta(alpha: number, beta: number, rng: () => number): number {\n const a = Math.max(1, alpha)\n const b = Math.max(1, beta)\n const x = sampleGamma(a, rng)\n const y = sampleGamma(b, rng)\n return x / (x + y)\n}\n\nfunction sampleGamma(shape: number, rng: () => number): number {\n // Marsaglia–Tsang for shape ≥ 1.\n const d = shape - 1 / 3\n const c = 1 / Math.sqrt(9 * d)\n while (true) {\n let x: number\n let v: number\n do {\n const u1 = rng() || 1e-12\n const u2 = rng() || 1e-12\n // Box-Muller for a normal sample.\n x = Math.sqrt(-2 * Math.log(u1)) * Math.cos(2 * Math.PI * u2)\n v = 1 + c * x\n } while (v <= 0)\n v = v * v * v\n const u = rng()\n if (u < 1 - 0.0331 * x ** 4) return d * v\n if (Math.log(u) < 0.5 * x * x + d * (1 - v + Math.log(v))) return d * v\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8EA,SAAgB,wBACd,cACA,gBACA,MACwB;CACxB,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,QAAQ,KAAK,gBAAgB;CACnC,MAAM,SAAS,KAAK;CAEpB,MAAM,0BAAU,IAAI,IAAsB;CAC1C,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,CAAC;EAC/B,IAAI,KAAK,EAAE,KAAK;EAChB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,YAAY,eAAe,KAAK,MAAM;EAC1C,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,UAAU,QAAQ,IAAI,CAAC,KAAK,CAAC;EACnC,MAAM,IAAI,QAAQ;EAClB,MAAM,OAAO,MAAM,IAAI,KAAM,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAClE,MAAM,WACJ,IAAI,IACA,gBACA,QAAQ,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI,KAAK;EAGnE,MAAM,SAAS,KAAK,KAAK,QAAQ,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,CAAC,CAAC;EACjE,OAAO;GAAE,WAAW,EAAE;GAAW,YAAY,EAAE;GAAY;GAAG;GAAM;GAAU;EAAO;CACvF,CAAC;CAGD,MAAM,aAAa,QAAQ,UAAU;CACrC,IAAI,cAAc,QAAQ;EACxB,MAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,SAAS,KAAK,IAAI,GAAG,UAAU,MAAM,CAAC,CAAC;EAC3E,OAAO,UAAU,KAAK,OAAO;GAC3B,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO;GACP,QAAQ,qCAAqC,EAAE,EAAE;EACnD,EAAE;CACJ;CACA,MAAM,YAAY,SAAS;CAC3B,MAAM,cAAc,UAAU,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC9D,OAAO,UAAU,KAAK,MAAM;EAC1B,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,SAAS;EAC5F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,QAAQ;GACf,QAAQ,YAAY,EAAE,SAAS,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE,SAAS,EAAE,KAAK,QAAQ,CAAC,EAAE;EACjF;CACF,CAAC;AACH;;;;;;;;;;;;AA6BA,SAAgB,mBACd,cACA,gBACA,MACwB;CACxB,MAAM,YAAY,KAAK,qBAAqB;CAC5C,MAAM,SAAS,KAAK,cAAc;CAClC,MAAM,QAAQ,KAAK,aAAa;CAChC,MAAM,MAAM,QACV,KAAK,MACL,aAAa,KAAK,gBAAgB,YAAY,KAAK,CACrD;CAEA,MAAM,0BAAU,IAAI,IAAkD;CACtE,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EAEvD,IADa,EAAE,QAAQ,EAAE,SAAS,WACxB,IAAI,UAAU;OACnB,IAAI,YAAY;EACrB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,QAAQ,eAAe,KAAK,MAAM;EACtC,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EACvD,MAAM,IAAI,SAAS,IAAI;EACvB,MAAM,IAAI,QAAQ,IAAI;EAEtB,MAAM,UAAU,WAAW,GAAG,GAAG,GAAG;EACpC,MAAM,WAAW,KAAK,IAAI,UAAU,SAAS;EAG7C,MAAM,WAAY,IAAI,MAAO,IAAI,MAAM,KAAK,IAAI,IAAI;EACpD,MAAM,QAAQ,KAAK,IAAI,KAAM,KAAK,KAAK,QAAQ,CAAC;EAChD,MAAM,SAAS,KAAK,IAAI,GAAG,WAAW,UAAU,EAAE;EAClD,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,GAAG,IAAI,SAAS,IAAI;GACpB;GACA;GACA;GACA;GACA;EACF;CACF,CAAC;CAED,MAAM,cAAc,MAAM,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC1D,OAAO,MAAM,KAAK,MAAM;EACtB,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,KAAK,MAAM;EAC9F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,KAAK,IAAI,GAAG,YAAY;GAC/B,QAAQ,QAAQ,EAAE,EAAE,QAAQ,CAAC,EAAE,GAAG,EAAE,EAAE,QAAQ,CAAC,EAAE,WAAW,EAAE,QAAQ,QAAQ,CAAC,EAAE,WAAW,UAAU;EACxG;CACF,CAAC;AACH;;AAGA,SAAgB,2BACd,MACA,OAAyD,CAAC,GACvC;CACnB,MAAM,YAAY,KAAK,iBAAiB;CACxC,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,MAAyB,CAAC;CAChC,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,CAAC,EAAE,YAAY;EAQnB,MAAM,QAAQ,cAAc,GAAG,aAAa,YAAY,QAAQ;EAChE,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EAC1D,IAAI,KAAK;GACP,WAAW,EAAE;GACb,YAAY,EAAE;GACd;GACA,MAAM,SAAS;EACjB,CAAC;CACH;CACA,OAAO;AACT;;;;;;AASA,SAAS,WAAW,OAAe,MAAc,KAA2B;CAC1E,MAAM,IAAI,KAAK,IAAI,GAAG,KAAK;CAC3B,MAAM,IAAI,KAAK,IAAI,GAAG,IAAI;CAC1B,MAAM,IAAI,YAAY,GAAG,GAAG;CAE5B,OAAO,KAAK,IADF,YAAY,GAAG,GACT;AAClB;AAEA,SAAS,YAAY,OAAe,KAA2B;CAE7D,MAAM,IAAI,QAAQ,IAAI;CACtB,MAAM,IAAI,IAAI,KAAK,KAAK,IAAI,CAAC;CAC7B,OAAO,MAAM;EACX,IAAI;EACJ,IAAI;EACJ,GAAG;GACD,MAAM,KAAK,IAAI,KAAK;GACpB,MAAM,KAAK,IAAI,KAAK;GAEpB,IAAI,KAAK,KAAK,KAAK,KAAK,IAAI,EAAE,CAAC,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;GAC5D,IAAI,IAAI,IAAI;EACd,SAAS,KAAK;EACd,IAAI,IAAI,IAAI;EACZ,MAAM,IAAI,IAAI;EACd,IAAI,IAAI,IAAI,QAAS,KAAK,GAAG,OAAO,IAAI;EACxC,IAAI,KAAK,IAAI,CAAC,IAAI,KAAM,IAAI,IAAI,KAAK,IAAI,IAAI,KAAK,IAAI,CAAC,IAAI,OAAO,IAAI;CACxE;AACF"}
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-Dy237wiH.js";
|
|
2
|
+
import "../index-Bfs5aufo.js";
|
|
3
|
+
//#region src/adapters/http.d.ts
|
|
4
|
+
interface HttpDispatchOptions<TScenario extends Scenario, _TArtifact> {
|
|
5
|
+
/** Static endpoint URL. Mutually exclusive with `resolveUrl`. */
|
|
6
|
+
url?: string;
|
|
7
|
+
/**
|
|
8
|
+
* Dynamic per-cell URL resolver. Receives the scenario + the substrate
|
|
9
|
+
* placement key (from `RunCampaignOptions.cellPlacement`) and returns the
|
|
10
|
+
* worker URL to invoke. Mutually exclusive with `url`.
|
|
11
|
+
*/
|
|
12
|
+
resolveUrl?: (input: {
|
|
13
|
+
scenario: TScenario;
|
|
14
|
+
placement?: string;
|
|
15
|
+
cellId: string;
|
|
16
|
+
}) => string;
|
|
17
|
+
/** Bearer token or static auth string set as `Authorization`. */
|
|
18
|
+
auth?: string | (() => string | Promise<string>);
|
|
19
|
+
/** Extra headers merged into every request. */
|
|
20
|
+
headers?: Record<string, string>;
|
|
21
|
+
/** Per-call timeout in ms. Default 5 minutes. */
|
|
22
|
+
timeoutMs?: number;
|
|
23
|
+
/** How many idempotent retries on 5xx / network errors. Default 2. */
|
|
24
|
+
retries?: number;
|
|
25
|
+
/** Optional fetch override (auth wrappers, custom agent, mocks). */
|
|
26
|
+
fetchImpl?: typeof fetch;
|
|
27
|
+
}
|
|
28
|
+
interface HttpDispatchRequestBody<TScenario extends Scenario> {
|
|
29
|
+
scenario: TScenario;
|
|
30
|
+
cellId: string;
|
|
31
|
+
runAttemptId: string;
|
|
32
|
+
rep: number;
|
|
33
|
+
generation?: number;
|
|
34
|
+
seed: number;
|
|
35
|
+
placement?: string;
|
|
36
|
+
cycleId?: string;
|
|
37
|
+
}
|
|
38
|
+
interface HttpDispatchResponseBody<TArtifact> {
|
|
39
|
+
artifact: TArtifact;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Wrap a remote HTTP endpoint as a `Dispatch`. The remote side should run
|
|
43
|
+
* `runDispatchServer` (or any service that speaks the same wire shape).
|
|
44
|
+
*
|
|
45
|
+
* Cancellation: the substrate's per-cell `AbortSignal` is forwarded; the
|
|
46
|
+
* server's `runDispatchServer` translates the resulting `AbortError` into
|
|
47
|
+
* a 499 (client-closed) so the client doesn't retry.
|
|
48
|
+
*/
|
|
49
|
+
declare function httpDispatch<TScenario extends Scenario, TArtifact>(opts: HttpDispatchOptions<TScenario, TArtifact>): DispatchFn<TScenario, TArtifact>;
|
|
50
|
+
interface RunDispatchServerOptions<TScenario extends Scenario, TArtifact> {
|
|
51
|
+
/** The Dispatch this server exposes — what runs when a request lands. */
|
|
52
|
+
dispatch: DispatchFn<TScenario, TArtifact>;
|
|
53
|
+
/** TCP port to bind. */
|
|
54
|
+
port: number;
|
|
55
|
+
/** Optional bind host; defaults to 0.0.0.0. */
|
|
56
|
+
host?: string;
|
|
57
|
+
/** Required for any non-test deployment: the bearer token clients must
|
|
58
|
+
* send. The substrate refuses to start without auth unless `auth: false`
|
|
59
|
+
* is set explicitly (intended ONLY for closed-network/internal testing). */
|
|
60
|
+
auth: string | false;
|
|
61
|
+
/** Path the server listens on. Default `/dispatch`. */
|
|
62
|
+
path?: string;
|
|
63
|
+
/**
|
|
64
|
+
* Per-request handler that wraps `dispatch` with whatever context the
|
|
65
|
+
* worker side needs to construct a `DispatchContext` — typically the
|
|
66
|
+
* trace writer, artifact writer, and cost meter. The substrate provides
|
|
67
|
+
* synthetic-but-typed defaults if not supplied; production deployments
|
|
68
|
+
* should wire real ones (e.g. ship traces to your OTel collector).
|
|
69
|
+
*/
|
|
70
|
+
contextFactory?: (req: HttpDispatchRequestBody<TScenario>, signal: AbortSignal) => Promise<DispatchContext>;
|
|
71
|
+
/** Optional max payload size for the request body (bytes). Default 10 MB. */
|
|
72
|
+
maxBodyBytes?: number;
|
|
73
|
+
/** Hook for observability — called on every successful or failed turn. */
|
|
74
|
+
onRequest?: (event: {
|
|
75
|
+
cellId: string;
|
|
76
|
+
durationMs: number;
|
|
77
|
+
success: boolean;
|
|
78
|
+
error?: unknown;
|
|
79
|
+
}) => void;
|
|
80
|
+
}
|
|
81
|
+
interface DispatchServerHandle {
|
|
82
|
+
/** The actual bound port (useful when `port: 0` requests an ephemeral port). */
|
|
83
|
+
port: number;
|
|
84
|
+
/** Stop accepting new connections and drain existing ones. */
|
|
85
|
+
close: () => Promise<void>;
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* Start an HTTP server exposing a local `Dispatch` over the wire. Pair with
|
|
89
|
+
* `httpDispatch` on the driver side.
|
|
90
|
+
*
|
|
91
|
+
* Wire shape:
|
|
92
|
+
*
|
|
93
|
+
* POST /dispatch
|
|
94
|
+
* Authorization: Bearer <token>
|
|
95
|
+
* Body: HttpDispatchRequestBody
|
|
96
|
+
* 200 OK: HttpDispatchResponseBody
|
|
97
|
+
* 401: missing/invalid auth
|
|
98
|
+
* 408: per-request timeout exceeded
|
|
99
|
+
* 499: client aborted before completion
|
|
100
|
+
* 500: dispatch threw
|
|
101
|
+
*
|
|
102
|
+
* The server is `node:http`-based to keep the runtime dependency surface
|
|
103
|
+
* minimal — works in plain Node, sandbox, or any container.
|
|
104
|
+
*/
|
|
105
|
+
declare function runDispatchServer<TScenario extends Scenario, TArtifact>(opts: RunDispatchServerOptions<TScenario, TArtifact>): Promise<DispatchServerHandle>;
|
|
106
|
+
//#endregion
|
|
107
|
+
export { DispatchServerHandle, HttpDispatchOptions, HttpDispatchRequestBody, HttpDispatchResponseBody, RunDispatchServerOptions, httpDispatch, runDispatchServer };
|
|
108
|
+
//# sourceMappingURL=http.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"http.d.ts","names":[],"sources":["../../src/adapters/http.ts"],"mappings":";;;UA0CiB,oBAAoB,kBAAkB,UAAU;;EAE/D;;;;;;EAMA,cAAc;IAAS,UAAU;IAAW;IAAoB;;;EAEhE,gCAAgC;;EAEhC,UAAU;;EAEV;;EAEA;;EAEA,mBAAmB;;UAGJ,wBAAwB,kBAAkB;EACzD,UAAU;EACV;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe,yBAAyB;EACxC,UAAU;;;;;;;;;;iBAiBI,aAAa,kBAAkB,UAAU,WACvD,MAAM,oBAAoB,WAAW,aACpC,WAAS,WAAW;UAqFN,yBAAyB,kBAAkB,UAAU;;EAEpE,UAAU,WAAS,WAAW;;EAE9B;;EAEA;;;;EAIA;;EAEA;;;;;;;;EAQA,kBACE,KAAK,wBAAwB,YAC7B,QAAQ,gBACL,QAAQ;;EAEb;;EAEA,aAAa;IACX;IACA;IACA;IACA;;;UAIa;;EAEf;;EAEA,aAAa;;;;;;;;;;;;;;;;;;;;iBAqBO,kBAAkB,kBAAkB,UAAU,WAClE,MAAM,yBAAyB,WAAW,aACzC,QAAQ"}
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
//#region src/adapters/http.ts
|
|
2
|
+
function resolveAuth(auth) {
|
|
3
|
+
if (!auth) return Promise.resolve(null);
|
|
4
|
+
if (typeof auth === "string") return Promise.resolve(auth);
|
|
5
|
+
return Promise.resolve(auth());
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* Wrap a remote HTTP endpoint as a `Dispatch`. The remote side should run
|
|
9
|
+
* `runDispatchServer` (or any service that speaks the same wire shape).
|
|
10
|
+
*
|
|
11
|
+
* Cancellation: the substrate's per-cell `AbortSignal` is forwarded; the
|
|
12
|
+
* server's `runDispatchServer` translates the resulting `AbortError` into
|
|
13
|
+
* a 499 (client-closed) so the client doesn't retry.
|
|
14
|
+
*/
|
|
15
|
+
function httpDispatch(opts) {
|
|
16
|
+
if (!opts.url && !opts.resolveUrl) throw new Error("httpDispatch: pass exactly one of `url` or `resolveUrl`.");
|
|
17
|
+
if (opts.url && opts.resolveUrl) throw new Error("httpDispatch: pass exactly one of `url` or `resolveUrl`, not both.");
|
|
18
|
+
const timeoutMs = opts.timeoutMs ?? 300 * 1e3;
|
|
19
|
+
const maxRetries = opts.retries ?? 2;
|
|
20
|
+
const f = opts.fetchImpl ?? ((...args) => fetch(...args));
|
|
21
|
+
return async (scenario, ctx) => {
|
|
22
|
+
const url = opts.url ?? opts.resolveUrl({
|
|
23
|
+
scenario,
|
|
24
|
+
placement: ctx.placement,
|
|
25
|
+
cellId: ctx.cellId
|
|
26
|
+
});
|
|
27
|
+
const authValue = await resolveAuth(opts.auth);
|
|
28
|
+
const body = {
|
|
29
|
+
scenario,
|
|
30
|
+
cellId: ctx.cellId,
|
|
31
|
+
runAttemptId: ctx.runAttemptId,
|
|
32
|
+
rep: ctx.rep,
|
|
33
|
+
generation: ctx.generation,
|
|
34
|
+
seed: ctx.seed,
|
|
35
|
+
placement: ctx.placement,
|
|
36
|
+
cycleId: ctx.cycleId
|
|
37
|
+
};
|
|
38
|
+
let lastError;
|
|
39
|
+
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
40
|
+
const ourTimeout = AbortSignal.timeout(timeoutMs);
|
|
41
|
+
const combinedSignal = AbortSignal.any([ctx.signal, ourTimeout]);
|
|
42
|
+
try {
|
|
43
|
+
const res = await f(url, {
|
|
44
|
+
method: "POST",
|
|
45
|
+
headers: {
|
|
46
|
+
"Content-Type": "application/json",
|
|
47
|
+
...authValue ? { Authorization: authValue.startsWith("Bearer ") ? authValue : `Bearer ${authValue}` } : {},
|
|
48
|
+
...opts.headers
|
|
49
|
+
},
|
|
50
|
+
body: JSON.stringify(body),
|
|
51
|
+
signal: combinedSignal
|
|
52
|
+
});
|
|
53
|
+
if (!res.ok) {
|
|
54
|
+
if (!(res.status >= 500 || res.status === 408 || res.status === 429) || attempt === maxRetries) {
|
|
55
|
+
const text = await res.text().catch(() => "");
|
|
56
|
+
throw new Error(`httpDispatch ${url} failed (${res.status}): ${text.slice(0, 500)}`);
|
|
57
|
+
}
|
|
58
|
+
await sleep(2 ** attempt * 200 + Math.random() * 200);
|
|
59
|
+
continue;
|
|
60
|
+
}
|
|
61
|
+
return (await res.json()).artifact;
|
|
62
|
+
} catch (err) {
|
|
63
|
+
if (ctx.signal.aborted) throw err;
|
|
64
|
+
lastError = err;
|
|
65
|
+
if (attempt === maxRetries) throw err;
|
|
66
|
+
await sleep(2 ** attempt * 200 + Math.random() * 200);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
throw lastError ?? /* @__PURE__ */ new Error("httpDispatch exhausted retries");
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
function sleep(ms) {
|
|
73
|
+
return new Promise((resolve) => {
|
|
74
|
+
const t = setTimeout(resolve, ms);
|
|
75
|
+
if (typeof t.unref === "function") t.unref();
|
|
76
|
+
});
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Start an HTTP server exposing a local `Dispatch` over the wire. Pair with
|
|
80
|
+
* `httpDispatch` on the driver side.
|
|
81
|
+
*
|
|
82
|
+
* Wire shape:
|
|
83
|
+
*
|
|
84
|
+
* POST /dispatch
|
|
85
|
+
* Authorization: Bearer <token>
|
|
86
|
+
* Body: HttpDispatchRequestBody
|
|
87
|
+
* 200 OK: HttpDispatchResponseBody
|
|
88
|
+
* 401: missing/invalid auth
|
|
89
|
+
* 408: per-request timeout exceeded
|
|
90
|
+
* 499: client aborted before completion
|
|
91
|
+
* 500: dispatch threw
|
|
92
|
+
*
|
|
93
|
+
* The server is `node:http`-based to keep the runtime dependency surface
|
|
94
|
+
* minimal — works in plain Node, sandbox, or any container.
|
|
95
|
+
*/
|
|
96
|
+
async function runDispatchServer(opts) {
|
|
97
|
+
if (opts.auth === void 0) throw new Error("runDispatchServer: 'auth' is required (pass a bearer-token string, or `auth: false` explicitly for a closed-network test deployment).");
|
|
98
|
+
const path = opts.path ?? "/dispatch";
|
|
99
|
+
const maxBytes = opts.maxBodyBytes ?? 10 * 1024 * 1024;
|
|
100
|
+
const expectedAuth = typeof opts.auth === "string" ? `Bearer ${opts.auth.replace(/^Bearer\s+/, "")}` : null;
|
|
101
|
+
const { createServer } = await import("node:http");
|
|
102
|
+
const server = createServer(async (req, res) => {
|
|
103
|
+
const start = Date.now();
|
|
104
|
+
let cellId = "unknown";
|
|
105
|
+
let success = false;
|
|
106
|
+
let errCaught;
|
|
107
|
+
try {
|
|
108
|
+
if (req.method !== "POST" || req.url?.split("?")[0] !== path) {
|
|
109
|
+
res.statusCode = 404;
|
|
110
|
+
res.end("not found");
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
if (expectedAuth) {
|
|
114
|
+
if (req.headers.authorization !== expectedAuth) {
|
|
115
|
+
res.statusCode = 401;
|
|
116
|
+
res.end("unauthorized");
|
|
117
|
+
return;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
const chunks = [];
|
|
121
|
+
let totalBytes = 0;
|
|
122
|
+
const aborter = new AbortController();
|
|
123
|
+
res.on("close", () => {
|
|
124
|
+
if (!res.writableEnded) aborter.abort();
|
|
125
|
+
});
|
|
126
|
+
for await (const chunk of req) {
|
|
127
|
+
const buf = chunk;
|
|
128
|
+
totalBytes += buf.length;
|
|
129
|
+
if (totalBytes > maxBytes) {
|
|
130
|
+
res.statusCode = 413;
|
|
131
|
+
res.end("payload too large");
|
|
132
|
+
return;
|
|
133
|
+
}
|
|
134
|
+
chunks.push(buf);
|
|
135
|
+
}
|
|
136
|
+
const body = JSON.parse(Buffer.concat(chunks).toString("utf8"));
|
|
137
|
+
if (typeof body.runAttemptId !== "string" || body.runAttemptId.trim().length === 0) throw new Error("runDispatchServer: request runAttemptId is required");
|
|
138
|
+
cellId = body.cellId;
|
|
139
|
+
const ctx = opts.contextFactory ? await opts.contextFactory(body, aborter.signal) : {
|
|
140
|
+
cellId: body.cellId,
|
|
141
|
+
runAttemptId: body.runAttemptId,
|
|
142
|
+
rep: body.rep,
|
|
143
|
+
generation: body.generation,
|
|
144
|
+
seed: body.seed,
|
|
145
|
+
signal: aborter.signal,
|
|
146
|
+
placement: body.placement,
|
|
147
|
+
cycleId: body.cycleId,
|
|
148
|
+
trace: NOOP_TRACE,
|
|
149
|
+
artifacts: NOOP_ARTIFACTS,
|
|
150
|
+
cost: NOOP_COST
|
|
151
|
+
};
|
|
152
|
+
if (ctx.runAttemptId !== body.runAttemptId) throw new Error("runDispatchServer: contextFactory must preserve request runAttemptId");
|
|
153
|
+
const responseBody = { artifact: await opts.dispatch(body.scenario, ctx) };
|
|
154
|
+
res.statusCode = 200;
|
|
155
|
+
res.setHeader("content-type", "application/json");
|
|
156
|
+
res.end(JSON.stringify(responseBody));
|
|
157
|
+
success = true;
|
|
158
|
+
} catch (err) {
|
|
159
|
+
errCaught = err;
|
|
160
|
+
if (err?.name === "AbortError") {
|
|
161
|
+
res.statusCode = 499;
|
|
162
|
+
res.end("client aborted");
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
res.statusCode = 500;
|
|
166
|
+
res.setHeader("content-type", "application/json");
|
|
167
|
+
res.end(JSON.stringify({ error: err instanceof Error ? err.message : String(err) }));
|
|
168
|
+
} finally {
|
|
169
|
+
opts.onRequest?.({
|
|
170
|
+
cellId,
|
|
171
|
+
durationMs: Date.now() - start,
|
|
172
|
+
success,
|
|
173
|
+
error: errCaught
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
});
|
|
177
|
+
await new Promise((resolve, reject) => {
|
|
178
|
+
server.once("error", reject);
|
|
179
|
+
server.listen(opts.port, opts.host ?? "0.0.0.0", () => resolve());
|
|
180
|
+
});
|
|
181
|
+
const addr = server.address();
|
|
182
|
+
return {
|
|
183
|
+
port: typeof addr === "object" && addr ? addr.port : opts.port,
|
|
184
|
+
close: () => new Promise((resolve, reject) => {
|
|
185
|
+
server.close((err) => err ? reject(err) : resolve());
|
|
186
|
+
})
|
|
187
|
+
};
|
|
188
|
+
}
|
|
189
|
+
const NOOP_TRACE = { span: () => ({
|
|
190
|
+
end: () => {},
|
|
191
|
+
setAttribute: () => {},
|
|
192
|
+
setStatus: () => {},
|
|
193
|
+
recordException: () => {},
|
|
194
|
+
addEvent: () => {}
|
|
195
|
+
}) };
|
|
196
|
+
const NOOP_ARTIFACTS = {
|
|
197
|
+
write: async () => void 0,
|
|
198
|
+
read: async () => void 0,
|
|
199
|
+
list: async () => []
|
|
200
|
+
};
|
|
201
|
+
const NOOP_COST = {
|
|
202
|
+
record: () => {},
|
|
203
|
+
total: () => 0
|
|
204
|
+
};
|
|
205
|
+
//#endregion
|
|
206
|
+
export { httpDispatch, runDispatchServer };
|
|
207
|
+
|
|
208
|
+
//# sourceMappingURL=http.js.map
|