@tangle-network/agent-eval 0.161.1 → 0.170.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +156 -0
- package/README.md +2 -0
- package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
- package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
- package/dist/adapters/http.d.ts +108 -0
- package/dist/adapters/http.d.ts.map +1 -0
- package/dist/adapters/http.js +208 -0
- package/dist/adapters/http.js.map +1 -0
- package/dist/analyst/index.d.ts +40 -70
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +18 -311
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
- package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
- package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
- package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
- package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
- package/dist/benchmark-C4wk_Sjr.js.map +1 -0
- package/dist/{benchmark-command-BDC3Gocz.js → benchmark-command-BA7qOdWw.js} +239 -253
- package/dist/benchmark-command-BA7qOdWw.js.map +1 -0
- package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
- package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +22 -8
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BSmOwskD.js → campaign-BeCbxFqs.js} +19 -18
- package/dist/campaign-BeCbxFqs.js.map +1 -0
- package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
- package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
- package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
- package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
- package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
- package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
- package/dist/cli.js +3 -3
- package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
- package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
- package/dist/{client-L9VVPkim.d.ts → client-_Fsa5c2_.d.ts} +4 -4
- package/dist/{client-L9VVPkim.d.ts.map → client-_Fsa5c2_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -703
- package/dist/contract/index.js +14 -14
- package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
- package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
- package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
- package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
- package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-ovxrOP0_.d.ts} +6 -6
- package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-ovxrOP0_.d.ts.map} +1 -1
- package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-Clj-8igZ.js} +27 -15
- package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-Clj-8igZ.js.map} +1 -1
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-DVJm8Xlh.d.ts} +7 -7
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-DVJm8Xlh.d.ts.map} +1 -1
- package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
- package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
- package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
- package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
- package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
- package/dist/emitter-DeQHiDMm.js.map +1 -0
- package/dist/{engine-Cu5qD5Fc.d.ts → engine-D12Rb6WB.d.ts} +7 -7
- package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-D12Rb6WB.d.ts.map} +1 -1
- package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-JDTeE6Pl.js} +5 -5
- package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
- package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
- package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +5 -5
- package/dist/experiment/index.js +11 -11
- package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
- package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-CQxylYeG.js} +4 -11
- package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
- package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-Cex8Da2i.js} +26 -12
- package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
- package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
- package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -3
- package/dist/fuzz.d.ts.map +1 -1
- package/dist/fuzz.js +4 -10
- package/dist/fuzz.js.map +1 -1
- package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Bn7_xWCv.d.ts} +111 -111
- package/dist/heldout-gate-Bn7_xWCv.d.ts.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/index-Bfs5aufo.d.ts +704 -0
- package/dist/index-Bfs5aufo.d.ts.map +1 -0
- package/dist/{index-CGtH1piv.d.ts → index-CM-SM00y.d.ts} +7 -38
- package/dist/index-CM-SM00y.d.ts.map +1 -0
- package/dist/{index-D-V8gCs_.d.ts → index-DBbivBNs.d.ts} +30 -22
- package/dist/index-DBbivBNs.d.ts.map +1 -0
- package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
- package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
- package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
- package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
- package/dist/index.d.ts +67 -37
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +47 -41
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
- package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
- package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
- package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
- package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
- package/dist/internal-BMFSR8Ns.js.map +1 -0
- package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
- package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
- package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
- package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
- package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
- package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
- package/dist/llm-client-BFMRpmqb.js.map +1 -0
- package/dist/{llm-judge-BhasIPFT.js → llm-judge-DbJdo8Nj.js} +101 -23
- package/dist/llm-judge-DbJdo8Nj.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/{matrix-eXKRMHnL.d.ts → matrix-BpI5Trmo.d.ts} +3 -3
- package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-BpI5Trmo.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +5 -4
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +14 -22
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
- package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +4 -4
- package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
- package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
- package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
- package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
- package/dist/pipelines/index.d.ts +7 -6
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +5 -20
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
- package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
- package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
- package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
- package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
- package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
- package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
- package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
- package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
- package/dist/{produced-state-DZ89riy5.js → produced-state-CtSIp5cQ.js} +5 -5
- package/dist/{produced-state-DZ89riy5.js.map → produced-state-CtSIp5cQ.js.map} +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-CkXSgKkF.d.ts} +3 -3
- package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-CkXSgKkF.d.ts.map} +1 -1
- package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
- package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
- package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
- package/dist/proposal-findings-bko3GGy-.js.map +1 -0
- package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CIRUardl.d.ts} +891 -891
- package/dist/provenance-CIRUardl.d.ts.map +1 -0
- package/dist/{query-CHmMP42p.js → query-BPGMVlbM.js} +46 -4
- package/dist/query-BPGMVlbM.js.map +1 -0
- package/dist/{query-DxPYqpmT.d.ts → query-Na5gEIGd.d.ts} +21 -4
- package/dist/query-Na5gEIGd.d.ts.map +1 -0
- package/dist/random-Dn5fPWkt.js +21 -0
- package/dist/random-Dn5fPWkt.js.map +1 -0
- package/dist/record-id-DUgsK5qp.js +17 -0
- package/dist/record-id-DUgsK5qp.js.map +1 -0
- package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
- package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
- package/dist/{release-confidence-DKfD2RYU.js → release-confidence-CzUHc4z4.js} +4 -4
- package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
- package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
- package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +6 -6
- package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
- package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
- package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
- package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
- package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-SkxYgT0x.js} +3 -3
- package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
- package/dist/rl.d.ts +8 -8
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +13 -18
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
- package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
- package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
- package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
- package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
- package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
- package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
- package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
- package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
- package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
- package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-I36eejJx.js} +3 -3
- package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
- package/dist/{sequential-rYW-Ophm.js → sequential-B51qAYE4.js} +4 -4
- package/dist/{sequential-rYW-Ophm.js.map → sequential-B51qAYE4.js.map} +1 -1
- package/dist/{server-BtFd4uzB.js → server-CCEnywOR.js} +27 -19
- package/dist/server-CCEnywOR.js.map +1 -0
- package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-B2R9C5aG.js} +11 -11
- package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-B2R9C5aG.js.map} +1 -1
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-DFS7QGpS.d.ts} +3 -3
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-DFS7QGpS.d.ts.map} +1 -1
- package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
- package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
- package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
- package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
- package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-B2DJ_82T.d.ts} +102 -36
- package/dist/store-tool-spans-B2DJ_82T.d.ts.map +1 -0
- package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-B9o6tU8f.js} +23 -19
- package/dist/store-tool-spans-B9o6tU8f.js.map +1 -0
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
- package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
- package/dist/{summary-report-BI5hUtvK.js → summary-report-Bgh8CpNK.js} +7 -7
- package/dist/{summary-report-BI5hUtvK.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
- package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
- package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
- package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
- package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-BnXlCJZQ.d.ts} +3 -3
- package/dist/tool-groups-BnXlCJZQ.d.ts.map +1 -0
- package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
- package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
- package/dist/{tool-waste-BqzmVdJk.js → tool-waste-CwGHzBzX.js} +4 -4
- package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +3 -3
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -4
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +60 -62
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +13 -22
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
- package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +4 -14
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-Bfk0uxRj.d.ts.map +1 -1
- package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
- package/dist/types-CiWITkGo.js.map +1 -0
- package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
- package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
- package/dist/{types-D4s7Z6nq.d.ts → types-Dy237wiH.d.ts} +3 -3
- package/dist/{types-D4s7Z6nq.d.ts.map → types-Dy237wiH.d.ts.map} +1 -1
- package/dist/{types-BPb2Kf_C2.d.ts → types-i21ccEkr.d.ts} +3 -3
- package/dist/types-i21ccEkr.d.ts.map +1 -0
- package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
- package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
- package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
- package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
- package/dist/wire/index.d.ts +23 -8
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/code-agent-intake.md +64 -0
- package/docs/concepts.md +1 -1
- package/docs/design/statistics-decisions.md +1 -1
- package/docs/distributed-driver.md +3 -6
- package/docs/public-api.md +122 -106
- package/docs/wire-protocol.md +5 -3
- package/package.json +9 -2
- package/dist/benchmark-BhT16ep9.js.map +0 -1
- package/dist/benchmark-command-BDC3Gocz.js.map +0 -1
- package/dist/campaign-BSmOwskD.js.map +0 -1
- package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
- package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
- package/dist/contract/index.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-DptEII26.js.map +0 -1
- package/dist/emitter-BpYFQPj4.js.map +0 -1
- package/dist/external-optimizer-process-WosTBChy.js.map +0 -1
- package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +0 -1
- package/dist/index-CGtH1piv.d.ts.map +0 -1
- package/dist/index-D-V8gCs_.d.ts.map +0 -1
- package/dist/index-vrJugRal.d.ts +0 -1
- package/dist/internal-BDHPCnjk.js.map +0 -1
- package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
- package/dist/kind-factory-DY8FdoXf.js.map +0 -1
- package/dist/llm-client-hgDieDNN.js.map +0 -1
- package/dist/llm-judge-BhasIPFT.js.map +0 -1
- package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
- package/dist/query-CHmMP42p.js.map +0 -1
- package/dist/query-DxPYqpmT.d.ts.map +0 -1
- package/dist/run-score-lDzV0X8j.js.map +0 -1
- package/dist/server-BtFd4uzB.js.map +0 -1
- package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
- package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
- package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
- package/dist/types-BI4fT3HN.js.map +0 -1
- package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
package/dist/rl.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"rl.js","names":[],"sources":["../src/rl/adaptation-eval.ts","../src/rl/compute-curves.ts","../src/rl/contamination.ts","../src/rl/rollout-input.ts","../src/rl/exporters.ts","../src/rl/dataset.ts","../src/rl/corpus.ts","../src/rl/off-policy.ts","../src/rl/predictive-validity-researcher.ts","../src/rl/preferences.ts","../src/rl/process-reward.ts","../src/rl/rl-campaign.ts","../src/rl/run-record-adapters.ts","../src/rl/sim-fidelity.ts","../src/rl/tournament.ts","../src/rl/verified-findings-dataset.ts"],"sourcesContent":["/**\n * Sample-efficient adaptation evaluation.\n *\n * For foundation-model-based agents, the load-bearing capability isn't\n * raw end-state performance — it's *how fast the agent reaches that\n * performance from cold start*. The same model with a worse prompt that\n * adapts in 5 demonstrations beats the same model with a better prompt\n * that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)\n * reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …\n * in-context examples or fine-tune steps.\n *\n * This module ships:\n *\n * 1. `runAdaptationCurve` — given a runner that takes k demonstrations\n * and returns a score, produce the (k, score) curve.\n * 2. `compareAdaptationCurves` — paired comparison across two policies.\n * Returns per-k delta with bootstrap CIs and an \"area-under-curve\"\n * summary statistic.\n * 3. `firstPassK` — for pass/fail evaluation, the minimum k at which\n * the policy reliably passes (≥ pass-rate threshold over reps).\n *\n * Use cases:\n * - Compare two prompt designs that have similar end-state performance\n * but different in-context efficiency.\n * - Decide between fine-tuning and prompting based on adaptation cost.\n * - Detect when a policy \"memorizes\" k=0 inputs vs. genuinely adapts.\n */\n\nimport { makeRng } from '../statistics/internal'\n\nexport interface AdaptationRunner<S> {\n /**\n * Runs the policy on `scenario` with `k` demonstrations. Returns a\n * scalar score in [0, 1]. The runner is responsible for any caching;\n * the harness calls it once per (scenario, k, rep) cell.\n */\n run(args: { scenario: S; k: number; rep: number }): Promise<number>\n}\n\nexport interface RunAdaptationCurveOptions<S> {\n scenarios: S[]\n /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */\n ks?: number[]\n /** Reps per (scenario, k) cell. Default 3. */\n reps?: number\n runner: AdaptationRunner<S>\n /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */\n passThreshold?: number\n}\n\nexport interface AdaptationPoint {\n k: number\n meanScore: number\n passRate: number\n std: number\n n: number\n /** Per-scenario means at this k. */\n perScenario: Array<{ scenarioId: string; meanScore: number; passes: number; total: number }>\n}\n\nexport interface AdaptationCurve {\n points: AdaptationPoint[]\n /**\n * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`\n * tested reaches it.\n */\n firstPassK: number | null\n /**\n * Area under the (k, meanScore) curve, normalized by max-k. A\n * single-number summary of \"how well does this policy adapt from\n * cold-start to fully-conditioned.\" Higher = better adapter.\n */\n adaptationArea: number\n}\n\nexport async function runAdaptationCurve<S extends { scenarioId?: string }>(\n opts: RunAdaptationCurveOptions<S>,\n): Promise<AdaptationCurve> {\n const ks = opts.ks ?? [0, 1, 2, 4, 8, 16]\n const reps = opts.reps ?? 3\n const passThreshold = opts.passThreshold ?? 0.5\n const sortedKs = [...ks].sort((a, b) => a - b)\n\n const points: AdaptationPoint[] = []\n for (const k of sortedKs) {\n const perScenario: AdaptationPoint['perScenario'] = []\n const allScores: number[] = []\n let totalPasses = 0\n let totalAttempts = 0\n for (const scenario of opts.scenarios) {\n const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`\n const scores: number[] = []\n let passes = 0\n for (let r = 0; r < reps; r++) {\n const score = await opts.runner.run({ scenario, k, rep: r })\n scores.push(score)\n if (score >= passThreshold) passes++\n allScores.push(score)\n if (score >= passThreshold) totalPasses++\n totalAttempts++\n }\n const meanS = scores.reduce((s, v) => s + v, 0) / scores.length\n perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length })\n }\n const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length)\n const variance =\n allScores.length < 2\n ? 0\n : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1)\n points.push({\n k,\n meanScore,\n passRate: totalPasses / Math.max(1, totalAttempts),\n std: Math.sqrt(variance),\n n: allScores.length,\n perScenario,\n })\n }\n\n const firstPassK = points.find((p) => p.passRate >= passThreshold)?.k ?? null\n const maxK = sortedKs[sortedKs.length - 1] ?? 1\n // Trapezoidal area under the (k, meanScore) curve, normalized by k-range.\n let area = 0\n for (let i = 1; i < points.length; i++) {\n const x1 = points[i - 1]!.k\n const x2 = points[i]!.k\n const y1 = points[i - 1]!.meanScore\n const y2 = points[i]!.meanScore\n area += ((y1 + y2) / 2) * (x2 - x1)\n }\n const adaptationArea = maxK === 0 ? 0 : area / maxK\n\n return { points, firstPassK, adaptationArea }\n}\n\nexport interface CompareCurvesResult {\n perK: Array<{\n k: number\n deltaMean: number\n aLow: number\n aHigh: number\n bLow: number\n bHigh: number\n }>\n areaDelta: number\n firstPassKDelta: number | null\n /** Verdict: 'a_better' | 'b_better' | 'similar'. */\n verdict: 'a_better' | 'b_better' | 'similar'\n /** Rationale, ready to render. */\n rationale: string\n}\n\n/**\n * Paired comparison of two adaptation curves. Per-k deltas with 95%\n * bootstrap CIs (constructed from each curve's `perScenario` per-k means\n * — the bootstrap unit is the scenario, not the rep).\n */\nexport function compareAdaptationCurves(\n a: AdaptationCurve,\n b: AdaptationCurve,\n opts: { confidence?: number; bootstrapResamples?: number; seed?: number } = {},\n): CompareCurvesResult {\n const conf = opts.confidence ?? 0.95\n const resamples = opts.bootstrapResamples ?? 500\n const rng = makeRng(\n opts.seed,\n a.points.flatMap((point) => point.perScenario.map((cell) => cell.meanScore)),\n b.points.flatMap((point) => point.perScenario.map((cell) => cell.meanScore)),\n )\n\n const perK: CompareCurvesResult['perK'] = []\n for (const ap of a.points) {\n const bp = b.points.find((p) => p.k === ap.k)\n if (!bp) continue\n const aMeans = ap.perScenario.map((s) => s.meanScore)\n const bMeans = bp.perScenario.map((s) => s.meanScore)\n const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng)\n const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng)\n perK.push({\n k: ap.k,\n deltaMean: ap.meanScore - bp.meanScore,\n aLow: aCi.low,\n aHigh: aCi.high,\n bLow: bCi.low,\n bHigh: bCi.high,\n })\n }\n\n const areaDelta = a.adaptationArea - b.adaptationArea\n const firstPassKDelta =\n a.firstPassK !== null && b.firstPassK !== null\n ? b.firstPassK - a.firstPassK // smaller k for a means a adapts faster (positive delta)\n : null\n\n // Composite verdict: positive area delta + most per-k deltas in same\n // direction → that side wins. Within ε of zero on both → similar.\n const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length)\n let verdict: CompareCurvesResult['verdict']\n if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = 'similar'\n else if (meanDelta > 0 && areaDelta > 0) verdict = 'a_better'\n else if (meanDelta < 0 && areaDelta < 0) verdict = 'b_better'\n else verdict = 'similar'\n\n const rationale =\n `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` +\n (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : '')\n\n return { perK, areaDelta, firstPassKDelta, verdict, rationale }\n}\n\n/** First k at which the curve's per-scenario pass rate reliably hits the threshold. */\nexport function firstPassK(curve: AdaptationCurve, threshold = 0.5): number | null {\n return curve.points.find((p) => p.passRate >= threshold)?.k ?? null\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction bootstrapMeanCi(\n xs: number[],\n resamples: number,\n confidence: number,\n rng: () => number,\n): { low: number; high: number } {\n if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 }\n const samples = new Array<number>(resamples)\n for (let b = 0; b < resamples; b++) {\n let sum = 0\n for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)]!\n samples[b] = sum / xs.length\n }\n samples.sort((a, b) => a - b)\n const alpha = 1 - confidence\n return {\n low: samples[Math.floor((alpha / 2) * resamples)]!,\n high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]!,\n }\n}\n","/**\n * Test-time compute scaling curves.\n *\n * The test-time-compute frontier paper (Snell et al. 2024) and the\n * subsequent o1-style scaling work both show that LLM-agent capability\n * is a function of the compute budget at inference, not just of the\n * training run. The right way to characterize a candidate is therefore\n * a *curve* — score at compute budgets {1×, 4×, 16×, …} — not a single\n * point.\n *\n * This module ships:\n *\n * 1. The compute-curve harness — `runComputeCurve(runner, budgets)` —\n * that evaluates one candidate at a sequence of compute budgets\n * and returns the (compute, score) curve.\n * 2. A best-of-N evaluator — `bestOfN(runner, n, scoreFn)` — the\n * simplest test-time-compute scaling primitive: sample N\n * independent rollouts, return the best.\n * 3. A self-consistency evaluator — `selfConsistency(runner, n)` —\n * the majority-vote variant of best-of-N for tasks with a small\n * categorical answer space.\n * 4. Pareto-frontier extraction over multiple candidates — given\n * (candidate, compute, score) tuples, return the set of\n * candidate-compute combinations that aren't dominated.\n *\n * Caveat: \"compute\" here is the caller's notion of a compute unit. For\n * agent eval that's typically wall-time × parallelism, or token budget,\n * or LLM-call count. We accept whatever the caller provides; the curve\n * is on whatever axis they pick.\n */\n\nimport { ValidationError } from '../errors'\n\nexport interface ComputeCurveBudget {\n /** Identifier — for the report. Common: '1x', '4x', '16x'. */\n id: string\n /** Numeric value on the chosen axis (tokens, calls, USD, ms — caller picks). */\n cost: number\n /** Free-form metadata (the caller can carry per-budget config). */\n meta?: Record<string, unknown>\n}\n\nexport interface ComputeCurvePoint {\n budgetId: string\n cost: number\n score: number\n /** Number of underlying samples used at this budget. */\n samples: number\n /** Optional spread / variance information. */\n std?: number\n /** Any extra metrics the runner returned. */\n metrics?: Record<string, number>\n}\n\nexport interface ComputeCurve {\n candidateId: string\n points: ComputeCurvePoint[]\n /** Rough exponent fit: score ≈ a + b * log(cost). Useful for \"how steep is the curve?\" */\n logSlope: number | null\n /** Best (highest-score) point on the curve. */\n best: ComputeCurvePoint\n}\n\nexport interface RunComputeCurveOptions {\n candidateId: string\n budgets: ComputeCurveBudget[]\n /**\n * Run the candidate at one budget. Returns the realized score plus\n * optional spread + extra metrics.\n */\n runAtBudget: (budget: ComputeCurveBudget) => Promise<{\n score: number\n samples: number\n std?: number\n metrics?: Record<string, number>\n }>\n}\n\nexport async function runComputeCurve(opts: RunComputeCurveOptions): Promise<ComputeCurve> {\n const points: ComputeCurvePoint[] = []\n for (const budget of opts.budgets) {\n const r = await opts.runAtBudget(budget)\n points.push({\n budgetId: budget.id,\n cost: budget.cost,\n score: r.score,\n samples: r.samples,\n std: r.std,\n metrics: r.metrics,\n })\n }\n const sorted = [...points].sort((a, b) => a.cost - b.cost)\n const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null\n const best = points.reduce((a, b) => (b.score > a.score ? b : a))\n return { candidateId: opts.candidateId, points: sorted, logSlope, best }\n}\n\nexport interface ComputeBestOfNOptions<O> {\n /** Number of independent samples to draw. */\n n: number\n /** Sampler — produces one rollout. */\n sample: (sampleIdx: number) => Promise<O>\n /** Score one rollout. */\n scoreFn: (rollout: O) => Promise<number> | number\n}\n\nexport interface ComputeBestOfNResult<O> {\n best: O\n bestScore: number\n scores: number[]\n meanScore: number\n /** Index of the best rollout, for diagnostics. */\n bestIndex: number\n}\n\n/** The simplest test-time scaling primitive. */\nexport async function bestOfN<O>(opts: ComputeBestOfNOptions<O>): Promise<ComputeBestOfNResult<O>> {\n if (opts.n <= 0) throw new ValidationError('bestOfN: n must be > 0')\n const rollouts: O[] = []\n const scores: number[] = []\n for (let i = 0; i < opts.n; i++) {\n const r = await opts.sample(i)\n rollouts.push(r)\n scores.push(await opts.scoreFn(r))\n }\n let bestIndex = 0\n for (let i = 1; i < scores.length; i++) if (scores[i]! > scores[bestIndex]!) bestIndex = i\n const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length\n return {\n best: rollouts[bestIndex]!,\n bestScore: scores[bestIndex]!,\n scores,\n meanScore,\n bestIndex,\n }\n}\n\nexport interface SelfConsistencyOptions<O> {\n n: number\n sample: (sampleIdx: number) => Promise<O>\n /** Extract the canonical answer key (string) from a rollout. */\n answerKey: (rollout: O) => string\n}\n\nexport interface SelfConsistencyResult<O> {\n /** Modal answer (the majority vote). */\n answer: string\n /** Fraction of samples voting for the modal answer in [0, 1]. */\n agreement: number\n /** Histogram of all answers. */\n histogram: Record<string, number>\n /** A representative rollout that voted for the modal answer. */\n representative: O\n /** All rollouts. */\n rollouts: O[]\n}\n\n/**\n * Self-consistency / majority-vote test-time scaling. For tasks with a\n * small categorical answer space (math problems, multiple choice).\n */\nexport async function selfConsistency<O>(\n opts: SelfConsistencyOptions<O>,\n): Promise<SelfConsistencyResult<O>> {\n if (opts.n <= 0) throw new ValidationError('selfConsistency: n must be > 0')\n const rollouts: O[] = []\n const histogram: Record<string, number> = {}\n for (let i = 0; i < opts.n; i++) {\n const r = await opts.sample(i)\n rollouts.push(r)\n const key = opts.answerKey(r)\n histogram[key] = (histogram[key] ?? 0) + 1\n }\n let answer = ''\n let max = -1\n for (const [k, v] of Object.entries(histogram)) {\n if (v > max) {\n max = v\n answer = k\n }\n }\n const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0]!\n return {\n answer,\n agreement: max / opts.n,\n histogram,\n representative,\n rollouts,\n }\n}\n\n/**\n * Pareto frontier over (candidate, compute, score) tuples. A point is on\n * the frontier iff no other point dominates it in both score (higher\n * better) and cost (lower better). Returns the frontier sorted ascending\n * by cost.\n */\nexport interface ParetoPointInput {\n candidateId: string\n budgetId: string\n cost: number\n score: number\n}\n\nexport function paretoFrontier(points: ParetoPointInput[]): ParetoPointInput[] {\n const onFrontier: ParetoPointInput[] = []\n for (const p of points) {\n const dominated = points.some(\n (q) =>\n q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score),\n )\n if (!dominated) onFrontier.push(p)\n }\n return onFrontier.sort((a, b) => a.cost - b.cost)\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction fitLogSlope(points: ComputeCurvePoint[]): number {\n // OLS slope of score on log(cost). Used as a single-number summary of\n // how much marginal compute helps. Positive = score improves with\n // compute; near-zero = capability ceiling reached.\n const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)))\n const ys = points.map((p) => p.score)\n const n = xs.length\n const mx = xs.reduce((s, x) => s + x, 0) / n\n const my = ys.reduce((s, y) => s + y, 0) / n\n let num = 0\n let den = 0\n for (let i = 0; i < n; i++) {\n num += (xs[i]! - mx) * (ys[i]! - my)\n den += (xs[i]! - mx) ** 2\n }\n return den === 0 ? 0 : num / den\n}\n","/**\n * Contamination probe — held-out perturbation tests.\n *\n * The bug class: once a benchmark scenario set is published, models train\n * on it, and your scores become invalid. SWE-Bench-Verified, GPQA, and\n * MMLU-Pro all exist because their predecessors got contaminated within\n * months. The right defense is to keep a held-out *perturbed* version of\n * every scenario — same task, slightly different surface — and check\n * whether scores diverge significantly. Genuine capability transfers; rote\n * memorization doesn't.\n *\n * This module ships the probe contract:\n *\n * 1. A `ScenarioPerturbation` strategy type — function that produces a\n * perturbed scenario from an original.\n * 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs\n * both halves and reports per-scenario score divergence + a global\n * contamination verdict via paired Wilcoxon.\n * 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,\n * `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the\n * task's structural difficulty while breaking surface memorization.\n *\n * The verdict is conservative: if the perturbed-vs-original score\n * difference is statistically significant (BH-adjusted p < 0.05) AND\n * the median drop is > 5 percentage points, we flag *contamination\n * suspected*. False positives are possible (the perturbation might\n * actually be harder); the default is to flag for review, not to\n * autoreject.\n */\n\nimport { ValidationError } from '../errors'\nimport { benjaminiHochberg, wilcoxonSignedRank } from '../statistics'\n\nexport type ScenarioPerturbationKind =\n | 'rename_variables'\n | 'shuffle_order'\n | 'paraphrase'\n | 'inject_irrelevant_clause'\n | 'custom'\n\nexport interface ScenarioPerturbation<S> {\n kind: ScenarioPerturbationKind\n /** Apply to one scenario, return its perturbed sibling. */\n apply: (scenario: S) => Promise<S> | S\n /** Optional id — for the report. */\n id?: string\n}\n\nexport interface ContaminationProbeInput<S> {\n /** Identity of every scenario. The probe's `runFingerprint` keys on these. */\n scenarioId: (s: S) => string\n /** Original scenarios. */\n originals: S[]\n /**\n * Either pre-computed perturbations (one per original, same order) OR a\n * `perturbation` strategy that synthesizes them on the fly.\n */\n perturbed?: S[]\n perturbation?: ScenarioPerturbation<S>\n /**\n * Run the policy/agent against one scenario and return a scalar score\n * in [0, 1]. The probe doesn't care what the policy is — that's the\n * caller's contract.\n */\n scoreFn: (s: S) => Promise<number>\n}\n\nexport interface ContaminationProbeOptions {\n /** Drop scores below this from the probe; treats partial failures separately. Default 0. */\n scoreFloor?: number\n /**\n * BH-FDR threshold for declaring contamination on each per-scenario\n * delta. Default 0.05.\n */\n fdr?: number\n /**\n * Minimum median per-scenario drop to flag global contamination. Default\n * 0.05 (5 percentage points). Smaller drops may be noise.\n */\n minMedianDrop?: number\n}\n\nexport interface ContaminationProbeReport {\n perScenario: Array<{\n scenarioId: string\n originalScore: number\n perturbedScore: number\n delta: number // perturbed - original (negative = drop)\n /** Per-scenario q-value (single-test BH for a single scenario). Mainly for display. */\n qValue: number\n }>\n /** Wilcoxon paired-test on the deltas. */\n pairedTest: { w: number; p: number }\n medianDelta: number\n meanDelta: number\n contaminationSuspected: boolean\n reason: string\n /** Number of scenarios processed. */\n n: number\n}\n\nexport async function runContaminationProbe<S>(\n input: ContaminationProbeInput<S>,\n opts: ContaminationProbeOptions = {},\n): Promise<ContaminationProbeReport> {\n const fdr = opts.fdr ?? 0.05\n const minMedianDrop = opts.minMedianDrop ?? 0.05\n const floor = opts.scoreFloor ?? 0\n\n if (!input.perturbed && !input.perturbation) {\n throw new ValidationError(\n 'runContaminationProbe: must supply either `perturbed` or `perturbation`.',\n )\n }\n const perturbed: S[] =\n input.perturbed ?? (await Promise.all(input.originals.map((s) => input.perturbation!.apply(s))))\n if (perturbed.length !== input.originals.length) {\n throw new ValidationError(\n `runContaminationProbe: perturbed length ${perturbed.length} ≠ originals ${input.originals.length}`,\n )\n }\n\n // Score both halves.\n const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)))\n const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)))\n\n const perScenario = input.originals.map((s, i) => ({\n scenarioId: input.scenarioId(s),\n originalScore: origScores[i]!,\n perturbedScore: pertScores[i]!,\n delta: pertScores[i]! - origScores[i]!,\n qValue: NaN,\n }))\n\n // Drop scenarios below the floor (partial failures we don't trust).\n const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor)\n if (valid.length < 4) {\n return {\n perScenario,\n pairedTest: { w: 0, p: 1 },\n medianDelta: 0,\n meanDelta: 0,\n contaminationSuspected: false,\n reason: `insufficient valid scenarios (n=${valid.length}, need ≥ 4)`,\n n: valid.length,\n }\n }\n\n const origValid = valid.map((p) => p.originalScore)\n const pertValid = valid.map((p) => p.perturbedScore)\n const pairedTest = wilcoxonSignedRank(origValid, pertValid)\n const deltas = valid.map((p) => p.delta)\n const sortedDeltas = [...deltas].sort((a, b) => a - b)\n const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)]!\n const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length\n\n // Per-scenario q-values via BH on a synthetic per-scenario p-value\n // (one-sample bootstrap; we use the absolute delta normalized by median\n // as a coarse signal — this is a display aid, the load-bearing test\n // is the global Wilcoxon).\n const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)))\n const { qValues } = benjaminiHochberg(pseudoP, fdr)\n for (let i = 0; i < valid.length; i++) {\n const v = valid[i]!\n const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId)\n if (idx >= 0) perScenario[idx]!.qValue = qValues[i]!\n }\n\n const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop\n const reason = contaminationSuspected\n ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} ≥ ${minMedianDrop}`\n : pairedTest.p >= fdr\n ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})`\n : `significant but small effect (median delta ${median.toFixed(4)})`\n\n return {\n perScenario,\n pairedTest,\n medianDelta: median,\n meanDelta: mean,\n contaminationSuspected,\n reason,\n n: valid.length,\n }\n}\n\n// ── Stock perturbations ──────────────────────────────────────────────────\n\n/**\n * Identifier-rename perturbation for code/text scenarios. Replaces every\n * occurrence of the listed identifiers with synthesized aliases. Use when\n * the scenario's structural difficulty is independent of variable names\n * (e.g. SWE-Bench-style coding tasks).\n */\nexport function renameVariables<S extends { prompt: string }>(\n identifiers: string[],\n rename: (name: string, idx: number) => string = (n, i) => `${n}_${((i % 26) + 10).toString(36)}`,\n): ScenarioPerturbation<S> {\n return {\n kind: 'rename_variables',\n apply(scenario) {\n let prompt = scenario.prompt\n identifiers.forEach((id, i) => {\n const replacement = rename(id, i)\n const re = new RegExp(`\\\\b${escapeRegex(id)}\\\\b`, 'g')\n prompt = prompt.replace(re, replacement)\n })\n return { ...scenario, prompt }\n },\n }\n}\n\n/**\n * Order-shuffle perturbation. Reshuffles a list-shaped section of the\n * prompt (for QA scenarios that present options A/B/C/D — answer depends\n * on the option labels, not order). Caller provides the section extractor.\n */\nexport function shuffleOrder<S extends { prompt: string }>(\n shuffleSection: (prompt: string, rng: () => number) => string,\n seed: number,\n): ScenarioPerturbation<S> {\n let s = seed >>> 0\n const rng = (): number => {\n s = (s + 0x6d2b79f5) >>> 0\n let t = s\n t = Math.imul(t ^ (t >>> 15), t | 1)\n t ^= t + Math.imul(t ^ (t >>> 7), t | 61)\n return ((t ^ (t >>> 14)) >>> 0) / 4294967296\n }\n return {\n kind: 'shuffle_order',\n apply(scenario) {\n const newPrompt = shuffleSection(scenario.prompt, rng)\n return { ...scenario, prompt: newPrompt }\n },\n }\n}\n\n/**\n * Inject-irrelevant-clause perturbation. Adds a benign sentence that\n * shouldn't change the answer. Tests for \"did the model just memorize\n * the input string.\"\n */\nexport function injectIrrelevantClause<S extends { prompt: string }>(\n clause: string,\n position: 'prefix' | 'suffix' = 'prefix',\n): ScenarioPerturbation<S> {\n return {\n kind: 'inject_irrelevant_clause',\n apply(scenario) {\n const prompt =\n position === 'prefix' ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`\n return { ...scenario, prompt }\n },\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction escapeRegex(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n}\n","/**\n * Shared checks for trainer exports over canonical minted rollout lines.\n *\n * Exporters accept only `MintedRolloutLine[]`. Callers convert run records with\n * `mintRolloutRows` before deriving preferences or trainer files.\n */\n\nimport { assertRewardGate, type MintedRolloutLine } from '../rollout/schema'\n\n/**\n * The line's reward, with `null` meaning \"no verdict exists\" and never \"scored\n * zero\".\n *\n * The wire contract represents an absent verdict directly as `reward: null`.\n */\nexport function trainableLineReward(line: MintedRolloutLine): number | null {\n // The single reward reader on the `rl/` side, so the runtime invariant check\n // sits here and covers GRPO rows, SFT row metadata, preference ordering, and\n // the published datasheet's reward distribution in one place.\n assertRewardGate(line, 'trainable reward')\n const { reward } = line.outcome\n if (reward === null || !Number.isFinite(reward)) return null\n return reward\n}\n\n/** The anti-Goodhart flag as it travels on the line. */\nexport function isLineRealnessGated(line: MintedRolloutLine): boolean {\n return line.outcome.realness_gated === true\n}\n\n/**\n * The minted lines behind a LINE-LESS training artifact.\n *\n * `PreferenceTriple`, `PrmTrainingTriple` and `StepReward` all carry a bare\n * reward number plus run ids, and nothing that says whether those runs faked\n * their success. An exporter over them therefore has no way, from its input\n * alone, to learn that its chosen side is a run the gate flagged — it will\n * happily emit the gaming trajectory as the preferred one. Supplying the lines\n * is what gives it eyes.\n */\nexport interface RolloutLineContext {\n /**\n * Minted lines for every INVOCATION the artifacts reference.\n *\n * Not \"one line per run\": `tangle.rollout.v1` models many invocations per\n * `run_id` — that is what `rollout_id` and `parent_rollout_id` are for, and\n * `supervisorRunRolloutLines` emits a supervisor node plus one per worker, all\n * sharing a single `run_id`. A reference is resolved against `rollout_id`\n * first and falls back to `run_id` only when that run has exactly one\n * invocation; see `resolveInvocation`.\n */\n lines: MintedRolloutLine[]\n}\n\n/** How one exporter names itself and its context type in the failure messages. */\nexport interface LineContextRequirement {\n /** Exporter label, e.g. `'DPO export'`. */\n exporter: string\n /** The context type the caller must pass, e.g. `'DpoLineContext'`. */\n contextType: string\n /** Why this exporter cannot see the gate without lines. One sentence. */\n because: string\n}\n\n/**\n * How ONE reference on a line-less artifact resolved against the supplied lines.\n *\n * `'ambiguous'` is a real answer, not an error path: the id named more than one\n * invocation and there is no way to tell which one the artifact meant.\n */\ntype Resolution =\n | { kind: 'resolved'; line: MintedRolloutLine }\n | { kind: 'ambiguous'; count: number }\n | { kind: 'missing' }\n\n/**\n * Both identity keys a line-less artifact might be carrying, built once.\n *\n * `rollout_id` is the INVOCATION id and is the key that answers the question\n * these exporters ask (\"was the run behind this side of the pair gamed?\").\n * `run_id` is an EPISODE id: `mintRolloutRows` writes it equal to `rollout_id`\n * for a solo run, but `supervisorRunRolloutLines` shares one `run_id` across a\n * supervisor node and every worker it spawned. The published artifacts\n * (`PreferenceTriple.chosenRunId`, `StepReward.runId`,\n * `PrmTrainingTriple.prefixRunId`) name run ids, so the fallback has to exist —\n * but only where it cannot be wrong.\n */\ninterface InvocationIndex {\n byRollout: Map<string, MintedRolloutLine[]>\n byRun: Map<string, MintedRolloutLine[]>\n}\n\nfunction push(index: Map<string, MintedRolloutLine[]>, key: string, line: MintedRolloutLine): void {\n const existing = index.get(key)\n if (existing === undefined) index.set(key, [line])\n else existing.push(line)\n}\n\nfunction invocationIndex(lines: readonly MintedRolloutLine[]): InvocationIndex {\n const byRollout = new Map<string, MintedRolloutLine[]>()\n const byRun = new Map<string, MintedRolloutLine[]>()\n for (const line of lines) {\n push(byRollout, line.rollout_id, line)\n push(byRun, line.run_id, line)\n }\n return { byRollout, byRun }\n}\n\n/**\n * Resolve one referenced id to exactly ONE invocation, or refuse to guess.\n *\n * The previous implementation was `new Map(lines.map((l) => [l.run_id, l]))`,\n * which is LAST-WINS: with a gated supervisor node and an ungated worker sharing\n * a `run_id`, the answer depended on which one appeared later in the array, so\n * `[gatedRoot, worker, rival]` emitted the gamed trajectory as `chosen` and\n * simply reordering the same input suppressed it. An order-dependent security\n * property passes every test whose fixture happens to be ordered favourably,\n * which is the worst possible failure mode for a gate.\n *\n * The rule that removes order from the answer: an id resolves only when it names\n * one invocation. `rollout_id` is tried first because it IS the invocation id;\n * `run_id` is accepted only when the run holds a single invocation, and a\n * cross-index disagreement (an id that is one line's `rollout_id` and a\n * different line's `run_id`) is ambiguous rather than silently preferring\n * either.\n */\nfunction resolveInvocation(index: InvocationIndex, id: string): Resolution {\n const rollouts = index.byRollout.get(id) ?? []\n const runs = index.byRun.get(id) ?? []\n if (rollouts.length > 1) return { kind: 'ambiguous', count: rollouts.length }\n const exact = rollouts[0]\n if (exact !== undefined) {\n if (runs.some((line) => line !== exact)) {\n return { kind: 'ambiguous', count: 1 + runs.filter((line) => line !== exact).length }\n }\n return { kind: 'resolved', line: exact }\n }\n if (runs.length > 1) return { kind: 'ambiguous', count: runs.length }\n const only = runs[0]\n return only === undefined ? { kind: 'missing' } : { kind: 'resolved', line: only }\n}\n\n/** What the admission rule did, item by item — the count a caller has to be able to see. */\nexport interface AdmissionAudit<T> {\n admitted: T[]\n /** Items dropped because a referenced invocation was realness-gated. */\n gatedDrops: number\n /** Items dropped because a referenced id named more than one invocation. */\n ambiguousDrops: number\n /** Each ambiguous id and how many invocations it named, deduped, first-seen order. */\n ambiguous: Array<{ id: string; invocations: number }>\n}\n\n/**\n * THE admission rule for every exporter whose input is line-less — one\n * implementation, because two siblings over the same input class with different\n * gating is the defect being eliminated, and it has now happened twice\n * (`toPrmRows` hardened while `toDpoRows` was left open; `toGrpoRows`'\n * `rewardOf` gated while `extractPreferences`' identically-named hook was not).\n *\n * Fail-closed in five steps:\n * 1. No context at all → throw. A two-argument call used to be accepted and\n * produced rows with no gate applied whatsoever.\n * 2. A referenced id with NO line → throw. Its gate status is unknown, and\n * unknown is not clean. Thrown rather than dropped because it means the\n * caller did not supply the context it was asked for, which is a defect in\n * the call, not in the data.\n * 3. A referenced id naming MORE THAN ONE invocation → DROP the item and count\n * it. Dropped rather than thrown because, unlike (2), this is ordinary data\n * — a supervision episode legitimately holds a supervisor invocation and\n * several workers under one `run_id` — and throwing would make these\n * exporters unusable on any supervisor corpus, whose only workaround is for\n * the caller to hand-filter `context.lines` down to one line per run. That\n * workaround IS the leak, performed by hand. The count is surfaced by\n * `admitUngatedByInvocation` so the drop is never silent.\n * 4. Every resolved line goes through `assertRewardGate`, so the line-less\n * exporters compose the same check list as the waist exporters instead of\n * relying on `realness_gated` alone (which is one of three checks).\n * 5. Either side realness-gated → DROP the item. Dropped rather than zeroed\n * because these shapes have no honest zero: a preference pair is a\n * statement that one trajectory is better than another, and a gamed\n * trajectory belongs on neither side of it — as the chosen one it teaches\n * the gaming move outright, and as the rejected one it still ships the\n * gaming trajectory's text into the training file as a contrast example\n * nobody asked for.\n *\n * `inspect` is the per-exporter extra check (PRM's trajectory-completeness\n * rules). It runs on every resolved line before any item is admitted, so the\n * whole batch fails before a single row is built.\n *\n * Pure: it reports what it dropped and prints nothing.\n */\nexport function auditInvocationAdmission<T>(\n items: readonly T[],\n idsOf: (item: T) => readonly string[],\n context: RolloutLineContext | undefined | null,\n requirement: LineContextRequirement,\n inspect?: (line: MintedRolloutLine) => void,\n): AdmissionAudit<T> {\n if (context === undefined || context === null) {\n throw new Error(\n `${requirement.exporter}: a ${requirement.contextType} is required — ${requirement.because} Pass \\`{ lines: (await mintRolloutRows(...)).rows }\\`.`,\n )\n }\n const index = invocationIndex(context.lines)\n const audit: AdmissionAudit<T> = {\n admitted: [],\n gatedDrops: 0,\n ambiguousDrops: 0,\n ambiguous: [],\n }\n for (const item of items) {\n const lines: MintedRolloutLine[] = []\n let ambiguous = false\n for (const id of idsOf(item)) {\n const resolution = resolveInvocation(index, id)\n if (resolution.kind === 'missing') {\n throw new Error(\n `${requirement.exporter}: no rollout line supplied for run ${id} — its realness gate and capture quality are unknown`,\n )\n }\n if (resolution.kind === 'ambiguous') {\n ambiguous = true\n if (!audit.ambiguous.some((entry) => entry.id === id)) {\n audit.ambiguous.push({ id, invocations: resolution.count })\n }\n continue\n }\n lines.push(resolution.line)\n }\n for (const line of lines) {\n assertRewardGate(line, requirement.exporter)\n inspect?.(line)\n }\n if (ambiguous) {\n audit.ambiguousDrops++\n continue\n }\n if (lines.some(isLineRealnessGated)) {\n audit.gatedDrops++\n continue\n }\n audit.admitted.push(item)\n }\n return audit\n}\n\n/**\n * `auditInvocationAdmission` for the exporters, which return rows and have\n * nowhere to put a count.\n *\n * The ambiguous drops are announced rather than swallowed: a caller who asked\n * for N pairs and silently received N-k has no way to notice that a chunk of\n * their preference data quietly evaporated, and \"the training set got smaller\n * for a reason nobody printed\" is the same class of invisible failure as the\n * gate that never ran. A gated drop is NOT announced — that one is the gate\n * doing exactly its job, on the population the caller already knows is flagged.\n */\nexport function admitUngatedByInvocation<T>(\n items: readonly T[],\n idsOf: (item: T) => readonly string[],\n context: RolloutLineContext | undefined | null,\n requirement: LineContextRequirement,\n inspect?: (line: MintedRolloutLine) => void,\n): T[] {\n const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect)\n if (audit.ambiguousDrops > 0) {\n const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(', ')\n console.warn(\n `[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more ` +\n 'than one invocation in the supplied lines, so the realness gate cannot be read for the ' +\n 'invocation the artifact meant. Reference the `rollout_id` instead of the `run_id`, or ' +\n 'supply a context holding one invocation per run.',\n )\n }\n return audit.admitted\n}\n","/**\n * Trainer-format exporters.\n *\n * agent-eval produces canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`,\n * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume\n * different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI\n * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own\n * JSONL conventions. Rather than ship N adapters, this module ships the\n * canonical formats most production pipelines accept and ergonomic helpers\n * for the rest.\n *\n * Shapes:\n * - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed\n * by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.\n * - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.\n * Consumed by prime-rl GRPO, verl, OpenRLHF.\n * - **SFT** — `{messages[]}` JSONL with chosen completion as the final\n * assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,\n * Anthropic finetuning.\n * - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.\n * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.\n *\n * Why ship this in agent-eval rather than a separate adapter package: the\n * canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`, etc.) are\n * agent-eval's contract; without first-party exporters consumers reverse-\n * engineer the mapping every release. The exporters codify it.\n *\n * The exporters take callbacks for any field that isn't on the canonical\n * artifact (specifically: prompt + completion text, since the package\n * stores only their hashes by design — full text is the consumer's\n * trace store / raw event log).\n *\n * Every exporter that produces a training row accepts canonical minted rollout\n * lines. Convert run records once with `mintRolloutRows`; downstream transforms\n * then share one reward, split, and authenticity contract.\n */\n\nimport { isSplitEligible } from '../rollout/exporters'\nimport { assertRewardGate, type MintedRolloutLine, type RolloutSplit } from '../rollout/schema'\nimport type { PreferenceTriple } from './preferences'\nimport type { PrmTrainingTriple, StepReward } from './process-reward'\nimport {\n admitUngatedByInvocation,\n isLineRealnessGated,\n type LineContextRequirement,\n type RolloutLineContext,\n trainableLineReward,\n} from './rollout-input'\n\nexport type { RolloutLineContext } from './rollout-input'\n\n// ── DPO / IPO / KTO ──────────────────────────────────────────────────────\n\nexport interface DpoLookups {\n /** Resolve the prompt text for a run (typically from a trace store / raw event sink). */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the assistant completion text for a run. */\n completionOf: (runId: string) => string | Promise<string>\n}\n\nexport interface DpoExportRow {\n prompt: string\n chosen: string\n rejected: string\n /** Carried-through margin. Some KTO / IPO variants use this. */\n margin?: number\n /** Free-form metadata for downstream filtering / sharding. */\n meta?: Record<string, unknown>\n}\n\n/** The minted lines for the runs a `PreferenceTriple` names on each side. */\nexport type DpoLineContext = RolloutLineContext\n\nconst DPO_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'DPO export',\n contextType: 'DpoLineContext',\n because:\n 'a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair — which is DPO trained to PREFER the gaming trajectory.',\n}\n\n/**\n * Convert preference triples to TRL-compatible DPO rows. The shape\n * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset\n * entry; every major DPO trainer accepts it.\n *\n * `context` is REQUIRED, and for the same reason it is required on the sibling\n * `toPrmRows`: a triple is a line-less artifact. It names two run ids and a\n * margin, and nothing on it says whether either run was flagged as gamed —\n * so a two-argument call applied NO gate at all and emitted the row verbatim,\n * reachable straight through the published bundle builder\n * (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).\n * Triples whose chosen or rejected side is realness-gated are dropped; a triple\n * naming a run with no supplied line is refused. See `admitUngatedByInvocation` for\n * why dropping, not zeroing, is the right disposition for a preference pair.\n */\nexport async function toDpoRows(\n triples: PreferenceTriple[],\n lookups: DpoLookups,\n context: DpoLineContext,\n): Promise<DpoExportRow[]> {\n const admitted = admitUngatedByInvocation(\n triples,\n (t) => [t.chosenRunId, t.rejectedRunId],\n context,\n DPO_CONTEXT_REQUIREMENT,\n )\n const out: DpoExportRow[] = []\n for (const t of admitted) {\n const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([\n Promise.resolve(lookups.promptOf(t.chosenRunId)),\n Promise.resolve(lookups.promptOf(t.rejectedRunId)),\n Promise.resolve(lookups.completionOf(t.chosenRunId)),\n Promise.resolve(lookups.completionOf(t.rejectedRunId)),\n ])\n if (chosenPrompt !== rejectedPrompt) {\n throw new Error(\n `toDpoRows: preference \"${t.chosenRunId}\"/\"${t.rejectedRunId}\" resolves to different prompts`,\n )\n }\n out.push({\n prompt: chosenPrompt,\n chosen,\n rejected,\n margin: t.marginScore,\n meta: {\n scenarioId: t.scenarioId,\n chosenVariantId: t.chosenVariantId,\n rejectedVariantId: t.rejectedVariantId,\n chosenRunId: t.chosenRunId,\n rejectedRunId: t.rejectedRunId,\n chosenModel: t.meta.chosenModel,\n rejectedModel: t.meta.rejectedModel,\n },\n })\n }\n return out\n}\n\n/** Serialize DPO rows as JSONL. One line per row. */\nexport function toDpoJsonl(rows: DpoExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── GRPO offline ─────────────────────────────────────────────────────────\n\nexport interface TrainingLineSelectionOptions {\n /** Include held-out evaluation data in training output. Default false. */\n allowHeldOutTrainingData?: boolean\n /** Require quality to be strictly greater than this value. Default 0. */\n minimumQualityExclusive?: number\n /**\n * Explicit split selection, replacing the default trainable-split rule.\n * Use this only when producing a deliberately named non-training slice.\n */\n splitFilter?: RolloutSplit[]\n}\n\nexport interface GrpoLookups\n extends Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'> {\n /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the assistant completion text for a rollout. */\n completionOf: (runId: string) => string | Promise<string>\n}\n\nexport interface GrpoExportRow {\n prompt: string\n completions: string[]\n rewards: number[]\n /** runIds in the same order as `completions[]` for traceability. */\n runIds: string[]\n meta?: Record<string, unknown>\n}\n\n/**\n * Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —\n * one row per scenario, with one completion per rollout on that scenario.\n * A scenario with fewer than two rewarded completions emits no row because a\n * group of one has no relative baseline.\n *\n * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages\n * within a group of completions for the same prompt; this is the\n * canonical input format. That relative baseline is exactly why the gate has\n * to hold here: one gamed sibling exporting at full reward shifts the advantage\n * of every honest run beside it.\n *\n * On the line path a realness-gated line stays in its group at reward 0 rather\n * than being dropped. 0 is the honest label for a faked success and is usable\n * signal; removing the line would also move the group's baseline, just in the\n * other direction. (SFT differs — see `toSftRows`.)\n */\nexport async function toGrpoRows(\n lines: MintedRolloutLine[],\n lookups: GrpoLookups,\n): Promise<GrpoExportRow[]> {\n return grpoRowsFromLines(lines, lookups)\n}\n\nasync function grpoRowsFromLines(\n lines: MintedRolloutLine[],\n lookups: GrpoLookups,\n): Promise<GrpoExportRow[]> {\n const grouped = new Map<string, MintedRolloutLine[]>()\n for (const line of lines) {\n if (!isSelectedSplit(line, lookups)) continue\n const arr = grouped.get(line.task.instance_id) ?? []\n arr.push(line)\n grouped.set(line.task.instance_id, arr)\n }\n\n const rows: GrpoExportRow[] = []\n for (const [scenarioId, group] of grouped.entries()) {\n if (group.length === 0) continue\n const scored: Array<{ line: MintedRolloutLine; reward: number }> = []\n for (const line of group) {\n const reward = trainableLineReward(line)\n if (reward === null) continue\n scored.push({ line, reward })\n }\n // GRPO's advantage is relative to the group mean, and a single completion\n // has no baseline.\n if (scored.length < 2) continue\n const prompts = await Promise.all(\n scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id))),\n )\n const prompt = prompts[0]!\n if (prompts.some((value) => value !== prompt)) {\n throw new Error(\n `toGrpoRows: scenario \"${scenarioId}\" resolves to different prompt text within one group`,\n )\n }\n const completions = await Promise.all(\n scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id))),\n )\n const rewards = scored.map(({ reward }) => reward)\n const runIds = scored.map(({ line }) => line.run_id)\n rows.push({\n prompt,\n completions,\n rewards,\n runIds,\n meta: {\n scenarioId,\n n: completions.length,\n meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length,\n },\n })\n }\n return rows\n}\n\nexport function toGrpoJsonl(rows: GrpoExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── SFT ──────────────────────────────────────────────────────────────────\n\nexport interface SftLookups extends TrainingLineSelectionOptions {\n /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the assistant completion text for a rollout. */\n completionOf: (runId: string) => string | Promise<string>\n /** Optional system message. Default omits. */\n systemOf?: (line: MintedRolloutLine) => string | null | undefined\n /** Extra filter on top of the realness gate (e.g., low score, failed cases). */\n include?: (line: MintedRolloutLine) => boolean\n /** Include held-out lines under the default split rule. Default false. */\n allowHeldOutTrainingData?: boolean\n}\n\nexport interface SftExportRow {\n messages: Array<{ role: 'system' | 'user' | 'assistant'; content: string }>\n meta?: Record<string, unknown>\n}\n\n/**\n * Convert rollout lines into Hugging Face / OpenAI / Anthropic-style\n * conversational SFT rows. By default every qualifying line becomes one row;\n * pass `include` to filter further (e.g., keep only `reward >= 0.8` for\n * rejection-sampling SFT).\n *\n * Realness-gated lines are dropped outright, not zeroed. SFT is imitation\n * learning: unlike GRPO, where a 0 reward teaches \"this trajectory was bad\",\n * every row here is a target to copy, so a gamed trajectory must not be in the\n * file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.\n *\n * The exporter is fail-closed on the split, same rule as\n * `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by\n * default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and\n * `canary` never pass the default rule. A non-training bundle that wants an\n * explicit slice (e.g. a holdout-only eval bundle) names it with\n * `splitFilter: ['holdout']` — explicit selection replaces the default rule.\n */\nexport async function toSftRows(\n lines: MintedRolloutLine[],\n lookups: SftLookups,\n): Promise<SftExportRow[]> {\n return sftRowsFromLines(lines, lookups)\n}\n\nasync function sftRowsFromLines(\n lines: MintedRolloutLine[],\n lookups: SftLookups,\n): Promise<SftExportRow[]> {\n const include = lookups.include ?? (() => true)\n const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0\n if (!Number.isFinite(minimumQualityExclusive)) {\n throw new Error('minimumQualityExclusive must be finite')\n }\n const rows: SftExportRow[] = []\n for (const line of lines) {\n // Checked BEFORE the drop, so this path fails loud on an impossible line\n // exactly like `rollout/exporters.toSftRows` does rather than quietly\n // filtering it as if it were an ordinary gated row.\n assertRewardGate(line, 'SFT export')\n if (isLineRealnessGated(line)) continue\n if (!isSelectedSplit(line, lookups)) continue\n const score = trainableLineReward(line)\n if (score === null || score <= minimumQualityExclusive) continue\n if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {\n continue\n }\n if (!include(line)) continue\n const system = lookups.systemOf?.(line)\n const [prompt, completion] = await Promise.all([\n Promise.resolve(lookups.promptOf(line.run_id)),\n Promise.resolve(lookups.completionOf(line.run_id)),\n ])\n const messages: SftExportRow['messages'] = []\n if (system) messages.push({ role: 'system', content: system })\n messages.push({ role: 'user', content: prompt })\n messages.push({ role: 'assistant', content: completion })\n rows.push({\n messages,\n meta: {\n runId: line.run_id,\n candidateId: line.candidate_id ?? null,\n scenarioId: line.task.instance_id,\n score,\n model: line.policy.model,\n },\n })\n }\n return rows\n}\n\nexport function toSftJsonl(rows: SftExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── PRM ──────────────────────────────────────────────────────────────────\n\nexport interface PrmLookups {\n /** Resolve the prompt text for a run. */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the trajectory step text for a (runId, spanId) pair. */\n stepTextOf: (runId: string, spanId: string) => string | Promise<string>\n /** Optional: sequence of prefix span ids leading up to the divergence. */\n prefixOf?: (runId: string, prefixStepIndex: number) => string[] | Promise<string[]>\n}\n\nexport interface PrmExportRow {\n prompt: string\n /** Span ids for the steps before divergence — caller resolves text via `stepTextOf`. */\n prefixSpanIds: string[]\n prefixStepText: string[]\n chosenStep: string\n rejectedStep: string\n chosenReward: number\n rejectedReward: number\n marginScore: number\n meta?: Record<string, unknown>\n}\n\nexport interface PrmLineContext extends RolloutLineContext {\n /**\n * The `maxSteps` cap the lines were minted with, if any.\n *\n * `mintRolloutRows` drops the MIDDLE of an over-long trajectory and leaves no\n * marker on the line, so a capped trajectory is indistinguishable from a\n * short one. Declaring the cap lets this exporter refuse any line sitting at\n * it — a process-reward model trained on a trajectory with a hole in it\n * learns credit assignment that never happened.\n */\n mintedWithMaxSteps?: number\n}\n\n/**\n * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`\n * callback resolves span text from the consumer's trace store.\n *\n * Every referenced run is checked against its minted line before any row is\n * emitted, and the export FAILS LOUD on a trajectory that was never fully\n * captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected\n * side is realness-gated are dropped instead: a capture defect is the caller's\n * mint configuration and must be fixed, whereas a gamed run is exactly the\n * condition the gate exists to filter.\n *\n * `context` is REQUIRED. A two-argument call used to be accepted and produced\n * rows with no gate applied at all — a `PrmTrainingTriple` carries a bare\n * `chosenReward` number and nothing that says which run it came from is honest,\n * so with no lines this exporter has no way to learn that its chosen step is a\n * step from a run that faked its success. It now throws: fail closed, because\n * the alternative is a process-reward model taught to prefer the gaming move at\n * the exact step the gaming happened.\n */\nexport async function toPrmRows(\n triples: PrmTrainingTriple[],\n lookups: PrmLookups,\n context: PrmLineContext,\n): Promise<PrmExportRow[]> {\n const admitted = admitPrmTriples(triples, context)\n const rows: PrmExportRow[] = []\n for (const t of admitted) {\n const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId))\n const prefixSpanIds = lookups.prefixOf\n ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex))\n : []\n const prefixStepText: string[] = []\n for (const spanId of prefixSpanIds) {\n prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)))\n }\n const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId))\n const rejectedStep = await Promise.resolve(\n lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId),\n )\n rows.push({\n prompt,\n prefixSpanIds,\n prefixStepText,\n chosenStep,\n rejectedStep,\n chosenReward: t.chosenReward,\n rejectedReward: t.rejectedReward,\n marginScore: t.marginScore,\n meta: {\n prefixRunId: t.prefixRunId,\n rejectedRunId: t.rejectedRunId,\n prefixStepIndex: t.prefixStepIndex,\n },\n })\n }\n return rows\n}\n\n/**\n * Refuse to build a process-reward row from a trajectory we do not fully have.\n *\n * PRM training assigns credit step by step, so a missing or silently shortened\n * step list is not degraded data — it is data about a trajectory that never\n * existed. Every condition below throws rather than filters, because each one\n * means the CALLER's capture or mint configuration is wrong.\n */\nfunction assertPrmTrainableLine(line: MintedRolloutLine, mintedWithMaxSteps?: number): void {\n const id = line.rollout_id\n if (line.provenance.gap !== undefined) {\n throw new Error(\n `PRM export: rollout ${id} is a gap line (${line.provenance.gap}) — refusing to build a process-reward row from a trajectory that was never captured`,\n )\n }\n if (line.steps === undefined || line.steps.length === 0) {\n throw new Error(\n `PRM export: rollout ${id} carries no steps — refusing to build a process-reward row with no trajectory`,\n )\n }\n if (line.outcome.is_truncated) {\n throw new Error(\n `PRM export: rollout ${id} is marked truncated — refusing to assign step-level credit over a partial trajectory`,\n )\n }\n if (mintedWithMaxSteps !== undefined && line.steps.length >= mintedWithMaxSteps) {\n throw new Error(\n `PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} — its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`,\n )\n }\n}\n\nconst PRM_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'PRM export',\n contextType: 'PrmLineContext',\n because:\n 'without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one.',\n}\n\n/**\n * Validate every referenced line up front (fail loud, before a single row is\n * written) and then drop the triples whose evidence is realness-gated.\n *\n * The gate half is `admitUngatedByInvocation`, shared with `toDpoRows` and\n * `stepRewardsToJsonl`; only the trajectory-completeness rules are PRM's own.\n */\nfunction admitPrmTriples(\n triples: PrmTrainingTriple[],\n context: PrmLineContext,\n): PrmTrainingTriple[] {\n return admitUngatedByInvocation(\n triples,\n (t) => [t.prefixRunId, t.rejectedRunId],\n context,\n PRM_CONTEXT_REQUIREMENT,\n (line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps),\n )\n}\n\nexport function toPrmJsonl(rows: PrmExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── Step rewards (for value-function regression) ─────────────────────────\n\nexport interface StepRewardJsonlRow {\n runId: string\n spanId: string\n stepIndex: number\n reward: number\n determinism: 'deterministic' | 'probabilistic'\n weight: number\n}\n\nconst STEP_REWARD_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'step-reward export',\n contextType: 'RolloutLineContext',\n because:\n 'a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success — so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere.',\n}\n\n/**\n * Step-level reward rows as JSONL.\n *\n * `context` is REQUIRED for the same reason it is on `toDpoRows` and\n * `toPrmRows`: this is a line-less input carrying a reward number. Steps\n * belonging to a realness-gated run are dropped rather than zeroed — a\n * per-step reward of 0 across a whole trajectory is a claim that every step was\n * bad, which is a different (and false) statement from \"this run's success was\n * fabricated, so its step-level credit assignment is meaningless\".\n */\nexport function stepRewardsToJsonl(stepRewards: StepReward[], context: RolloutLineContext): string {\n const admitted = admitUngatedByInvocation(\n stepRewards,\n (s) => [s.runId],\n context,\n STEP_REWARD_CONTEXT_REQUIREMENT,\n )\n const rows: StepRewardJsonlRow[] = admitted.map((s) => ({\n runId: s.runId,\n spanId: s.spanId,\n stepIndex: s.stepIndex,\n reward: s.reward,\n determinism: s.determinism,\n weight: s.weight ?? 1,\n }))\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\nfunction isSelectedSplit(\n line: MintedRolloutLine,\n options: Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'>,\n): boolean {\n if (options.splitFilter !== undefined) return options.splitFilter.includes(line.task.split)\n return isSplitEligible(line, options)\n}\n","/**\n * RL dataset packaging + datasheet — the publishable, sellable bundle.\n *\n * The format exporters (`toGrpoRows` / `toSftRows` / `toDpoRows`) already\n * produce trainer-ready shapes (prime-rl GRPO, TRL DPO, conversational SFT).\n * What turns that into a dataset someone can PUBLISH or BUY is the provenance\n * + a datasheet: which models produced it, which prompt/agent versions, how the\n * reward was derived (deterministic verifiable vs probabilistic judge — the\n * credibility axis a buyer checks first), the split discipline, the reward\n * distribution, the quality gates, the license, and the intended/out-of-scope\n * uses. This module computes those facts from `MintedRolloutLine[]` and renders a\n * \"Datasheet for Datasets\" (Gebru et al. 2018) card alongside the format files.\n *\n * It composes the existing `rl/exporters` — it does not reimplement any trainer\n * format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization\n * with per-token loss masks) is a downstream Python stage that consumes the\n * `messages`/`completions` this bundle emits.\n *\n * Input discipline: the bundle is built from `MintedRolloutLine[]`. The datasheet's\n * reward distribution ships INSIDE the published artifact, so it has to be\n * derived from exactly the same gated number as the rows it describes —\n * otherwise the provenance a buyer checks first is a lie. Taking the same\n * gated input as the exporters is what guarantees that.\n */\n\nimport type { MintedRolloutLine, RolloutSplit } from '../rollout/schema'\nimport {\n type DpoLookups,\n type GrpoLookups,\n type SftLookups,\n toDpoJsonl,\n toDpoRows,\n toGrpoJsonl,\n toGrpoRows,\n toSftJsonl,\n toSftRows,\n} from './exporters'\nimport type { PreferenceTriple } from './preferences'\nimport { trainableLineReward } from './rollout-input'\n\nexport type RewardKind = 'deterministic' | 'probabilistic' | 'mixed'\nconst DATASET_FORMATS = ['grpo', 'sft', 'dpo'] as const\nexport type DatasetFormat = (typeof DATASET_FORMATS)[number]\n\nconst DATASET_FORMAT_SET = new Set<unknown>(DATASET_FORMATS)\n\nexport function validateDatasetFormats(value: unknown): DatasetFormat[] {\n if (!Array.isArray(value) || value.length === 0) {\n throw new Error('buildRlDataset: formats must contain at least one of: grpo, sft, dpo')\n }\n\n const formats: DatasetFormat[] = []\n const seen = new Set<DatasetFormat>()\n for (const format of value) {\n if (!DATASET_FORMAT_SET.has(format)) {\n throw new Error(\n `buildRlDataset: unsupported format ${JSON.stringify(format)}; expected exactly one of: grpo, sft, dpo`,\n )\n }\n const datasetFormat = format as DatasetFormat\n if (seen.has(datasetFormat)) {\n throw new Error(\n `buildRlDataset: duplicate format ${JSON.stringify(datasetFormat)}; each format may be requested once`,\n )\n }\n seen.add(datasetFormat)\n formats.push(datasetFormat)\n }\n return formats\n}\n\n/** Caller-declared context — the qualitative half of the datasheet that can't\n * be computed from records. */\nexport interface RlDatasetConfig {\n name: string\n version: string\n /** Product/task domain, e.g. 'legal-m&a', 'tax-1040'. */\n domain: string\n /** SPDX id or a named commercial license. Required — an unlicensed dataset\n * cannot be published or sold. */\n license: string\n /** How the reward was produced. `kind: 'deterministic'` (a test/schema/XPath\n * decided it) is the credibility signal; 'probabilistic' = LLM-judge. */\n reward: { kind: RewardKind; source: string; description: string }\n intendedUse: string\n outOfScope: string\n limitations: string\n /** ISO timestamp — passed in (the substrate forbids Date.now()). */\n createdAtIso: string\n /** Default: ['sft']. GRPO must be requested for multi-completion groups. */\n formats?: DatasetFormat[]\n /** Quality gates already run, recorded on the card for the buyer. */\n qualityGates?: {\n contaminationProbe?: 'passed' | 'failed' | 'not-run'\n dedup?: boolean\n verifiableRewardFilter?: boolean\n }\n}\n\nexport interface RewardStats {\n n: number\n mean: number | null\n median: number | null\n min: number | null\n max: number | null\n std: number | null\n}\n\nexport interface RlDatasetStats {\n records: number\n /** Rollouts carrying an explicit task-quality score. */\n scoredRecords: number\n /** Rollout count per split. */\n splits: Record<RolloutSplit, number>\n reward: RewardStats\n /** Distinct snapshot-pinned models that produced the trajectories. */\n models: string[]\n /** Distinct effective-prompt hashes (the agent profile/prompt versions). */\n promptHashes: string[]\n commitShas: string[]\n totalTokens: { input: number; output: number }\n totalCostUsd: number\n /**\n * Rollouts whose USD cost was never captured (`cost.usd === null`). When\n * non-zero, `totalCostUsd` is a floor, not the bill — a published dataset\n * must not present an unbilled run as a $0 one.\n */\n rolloutsWithoutCost: number\n}\n\nexport interface RlDatasetManifest extends RlDatasetConfig {\n formats: DatasetFormat[]\n rowCounts: Partial<Record<DatasetFormat, number>>\n stats: RlDatasetStats\n}\n\nexport interface RlDatasetBundle {\n manifest: RlDatasetManifest\n /** Relative filename -> contents. Write these to a directory to publish. */\n files: Record<string, string>\n}\n\nfunction distinct(xs: Array<string | null | undefined>): string[] {\n return [...new Set(xs.filter((x): x is string => typeof x === 'string' && x.length > 0))].sort()\n}\n\nfunction computeRewardStats(values: number[]): RewardStats {\n if (values.length === 0) {\n return { n: 0, mean: null, median: null, min: null, max: null, std: null }\n }\n const sorted = [...values].sort((a, b) => a - b)\n const n = sorted.length\n const mean = sorted.reduce((s, x) => s + x, 0) / n\n const mid = Math.floor(n / 2)\n const median = n % 2 === 0 ? (sorted[mid - 1]! + sorted[mid]!) / 2 : sorted[mid]!\n const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n\n return { n, mean, median, min: sorted[0]!, max: sorted[n - 1]!, std: Math.sqrt(variance) }\n}\n\nfunction computeStatsFromLines(lines: MintedRolloutLine[]): RlDatasetStats {\n const splits: Record<RolloutSplit, number> = { search: 0, dev: 0, holdout: 0, canary: 0 }\n let inTok = 0\n let outTok = 0\n let cost = 0\n let rolloutsWithoutCost = 0\n const rewards: number[] = []\n for (const line of lines) {\n splits[line.task.split] += 1\n inTok += line.cost.tokens_in ?? 0\n outTok += line.cost.tokens_out ?? 0\n if (line.cost.usd === null) rolloutsWithoutCost++\n else cost += line.cost.usd\n const rw = trainableLineReward(line)\n if (rw !== null) rewards.push(rw)\n }\n return {\n records: lines.length,\n scoredRecords: rewards.length,\n splits,\n reward: computeRewardStats(rewards),\n models: distinct(lines.map((l) => l.policy.model)),\n promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),\n commitShas: distinct(lines.map((l) => l.policy.profile_commit)),\n totalTokens: { input: inTok, output: outTok },\n totalCostUsd: cost,\n rolloutsWithoutCost,\n }\n}\n\n/**\n * Package graded rollout lines into a publishable RL dataset bundle: the\n * trainer-format JSONL files + a manifest + a datasheet. DPO requires\n * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from\n * the lines directly via the supplied lookups. Throws on an empty corpus —\n * an empty dataset must never be published.\n */\nexport async function buildRlDataset(\n lines: MintedRolloutLine[],\n lookups: GrpoLookups & SftLookups,\n config: RlDatasetConfig,\n preferences?: { triples: PreferenceTriple[]; lookups: DpoLookups },\n): Promise<RlDatasetBundle> {\n if (lines.length === 0) {\n throw new Error('buildRlDataset: no rollout lines — refusing to package an empty dataset')\n }\n const formats = validateDatasetFormats(config.formats === undefined ? ['sft'] : config.formats)\n const files: Record<string, string> = {}\n const rowCounts: Partial<Record<DatasetFormat, number>> = {}\n\n if (formats.includes('grpo')) {\n const rows = await toGrpoRows(lines, lookups)\n requireRows('grpo', rows.length)\n files['train.grpo.jsonl'] = toGrpoJsonl(rows)\n rowCounts.grpo = rows.length\n }\n if (formats.includes('sft')) {\n const rows = await toSftRows(lines, lookups)\n requireRows('sft', rows.length)\n files['train.sft.jsonl'] = toSftJsonl(rows)\n rowCounts.sft = rows.length\n }\n if (formats.includes('dpo')) {\n if (!preferences) {\n throw new Error(\"buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)\")\n }\n const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines })\n requireRows('dpo', rows.length)\n files['train.dpo.jsonl'] = toDpoJsonl(rows)\n rowCounts.dpo = rows.length\n }\n if (!Object.keys(files).some((name) => name.startsWith('train.') && name.endsWith('.jsonl'))) {\n throw new Error('buildRlDataset: no trainer file was emitted')\n }\n\n const manifest: RlDatasetManifest = {\n ...config,\n formats,\n rowCounts,\n stats: computeStatsFromLines(lines),\n }\n files['manifest.json'] = `${JSON.stringify(manifest, null, 2)}\\n`\n files['DATASHEET.md'] = datasheetToMarkdown(manifest)\n return { manifest, files }\n}\n\nfunction requireRows(format: DatasetFormat, rows: number): void {\n if (rows === 0) {\n throw new Error(`buildRlDataset: requested '${format}' format produced no trainable rows`)\n }\n}\n\nfunction pct(x: number): string {\n return `${(x * 100).toFixed(1)}%`\n}\n\nfunction stat(value: number | null): string {\n return value === null ? 'n/a' : value.toFixed(3)\n}\n\n/** Render the \"Datasheet for Datasets\" card that a buyer reads. */\nexport function datasheetToMarkdown(m: RlDatasetManifest): string {\n const s = m.stats\n const total = s.records || 1\n const splitLines = (['search', 'dev', 'holdout', 'canary'] as RolloutSplit[])\n .map((k) => ` - \\`${k}\\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`)\n .join('\\n')\n const costNote =\n s.rolloutsWithoutCost > 0\n ? ` (floor — ${s.rolloutsWithoutCost} rollout(s) never captured a cost)`\n : ''\n const deterministic = m.reward.kind === 'deterministic'\n return [\n `# Dataset: ${m.name} \\`v${m.version}\\``,\n '',\n `**Domain:** ${m.domain} | **Created:** ${m.createdAtIso} | **License:** ${m.license}`,\n '',\n '## Reward provenance',\n `- **Kind:** ${m.reward.kind}${deterministic ? ' (decidable, not judge noise)' : ''}`,\n `- **Source:** ${m.reward.source}`,\n `- **Description:** ${m.reward.description}`,\n '',\n '## Composition',\n `- **Records (trajectories):** ${s.records}`,\n `- **Scored records:** ${s.scoredRecords}`,\n `- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(', ')}`,\n '- **Splits:**',\n splitLines,\n '',\n '## Reward distribution',\n `- n=${s.reward.n} | mean=${stat(s.reward.mean)} | median=${stat(s.reward.median)} | min=${stat(s.reward.min)} | max=${stat(s.reward.max)} | std=${stat(s.reward.std)}`,\n '',\n '## Provenance',\n `- **Models:** ${s.models.join(', ')}`,\n `- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,\n `- **Commits:** ${s.commitShas.join(', ')}`,\n `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,\n '',\n '## Quality gates',\n `- Contamination probe: ${m.qualityGates?.contaminationProbe ?? 'not-run'}`,\n `- Dedup: ${m.qualityGates?.dedup ? 'yes' : 'no'} | Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? 'yes' : 'no'}`,\n '',\n '## Recommended uses',\n m.intendedUse,\n '',\n '## Out of scope',\n m.outOfScope,\n '',\n '## Limitations',\n m.limitations,\n '',\n '## Token rendering',\n 'For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns. See `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.',\n '',\n ].join('\\n')\n}\n","/**\n * RL corpus — the durable, append-only accumulation of graded RunRecords that\n * every eval run deposits BY DEFAULT.\n *\n * The dataset is the free exhaust of the normal eval process: we run evals\n * constantly to get an agent production-ready, and those runs already produce\n * graded trajectories. Instead of writing them to an ephemeral run dir and\n * throwing them away, `appendToCorpus` accumulates them into a durable corpus;\n * `buildDatasetFromCorpus` later harvests the whole corpus into a publishable\n * bundle. No separate data-collection campaign — the data accrues from work we\n * do anyway. This is the \"best things for free by our process\" layer.\n *\n * Trajectory text rides on the record as top-level `prompt` / `completion`\n * (what the eval harnesses capture; the RunRecord validator ignores the extra\n * keys). The harvest reads them directly — no trace store round-trip needed.\n */\n\nimport { appendFileSync, existsSync, mkdirSync, readFileSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { mintRolloutRows } from '../rollout/mint'\nimport { trainingScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { InMemoryTraceStore } from '../trace/store'\nimport { buildRlDataset, type RlDatasetBundle, type RlDatasetConfig } from './dataset'\n\n/** A corpus record is a RunRecord carrying the trajectory text the harness\n * captured. `prompt`/`completion` are top-level (the validator ignores extras). */\nexport type CorpusRecord = RunRecord & { prompt?: string; completion?: string }\n\nexport interface CorpusAppendResult {\n appended: number\n /** Skipped because a record with the same runId was already in the corpus\n * (idempotent appends — NOT re-run collapsing; re-runs get fresh runIds). */\n skipped: number\n total: number\n}\n\n/**\n * Append graded records to the corpus (append-only JSONL). Deduplicates by\n * `runId` against what's already on disk so re-running the same harness is\n * idempotent. Creates the file and parent dir. This is the call every eval\n * harness makes by default after producing its records.\n */\nexport function appendToCorpus(records: CorpusRecord[], corpusPath: string): CorpusAppendResult {\n mkdirSync(dirname(corpusPath), { recursive: true })\n const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : []\n const seen = new Set(existing.map((r) => r.runId))\n const lines: string[] = []\n let appended = 0\n let skipped = 0\n for (const r of records) {\n if (seen.has(r.runId)) {\n skipped++\n continue\n }\n seen.add(r.runId)\n lines.push(JSON.stringify(r))\n appended++\n }\n if (lines.length > 0) appendFileSync(corpusPath, `${lines.join('\\n')}\\n`)\n return { appended, skipped, total: existing.length + appended }\n}\n\n/** Read the full corpus. Returns [] if the corpus does not exist yet. */\nexport function readCorpus(corpusPath: string): CorpusRecord[] {\n if (!existsSync(corpusPath)) return []\n const out: CorpusRecord[] = []\n for (const line of readFileSync(corpusPath, 'utf8').split('\\n')) {\n if (line.trim()) out.push(JSON.parse(line) as CorpusRecord)\n }\n return out\n}\n\n/**\n * The harvest's score reader is GATED: a gamed run reads 0, so it cannot buy\n * its way past `minScore` into the published bundle with its claimed score.\n * `null` = unscored (a labeled gap, dropped before packaging, never a 0).\n */\nfunction rewardOf(r: CorpusRecord): number | null {\n const v = trainingScore(r)\n return typeof v === 'number' && Number.isFinite(v) ? v : null\n}\n\nexport interface HarvestOptions {\n /** Keep only records scoring >= this (rejection-sampling for SFT). */\n minScore?: number\n /** Keep only these source splits. Held-out rows still require the explicit override below. */\n splits?: RunRecord['splitTag'][]\n /** Permit held-out rows in training files. Default false. */\n allowHeldOutTrainingData?: boolean\n}\n\n/**\n * Harvest the accumulated corpus into a publishable RL dataset bundle. Reads\n * trajectory text from each record's top-level `prompt`/`completion`; records\n * missing either are excluded (a graded score with no trajectory can't train).\n * Optionally filters by score / split. Throws (via buildRlDataset) if nothing\n * survives — an empty dataset must never be published.\n *\n * `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run\n * cannot buy its way into the published bundle with its claimed score —\n * `minScore` is exactly the door a reward-hacked run would otherwise clear for\n * SFT. Unscored records are dropped before packaging: a missing label is not a\n * zero, and it is not publishable either.\n */\nexport async function buildDatasetFromCorpus(\n corpusPath: string,\n config: RlDatasetConfig,\n opts: HarvestOptions = {},\n): Promise<RlDatasetBundle> {\n let records = readCorpus(corpusPath).filter(\n (r) => typeof r.prompt === 'string' && typeof r.completion === 'string',\n )\n if (opts.splits) records = records.filter((r) => opts.splits!.includes(r.splitTag))\n records = records.filter((r) => rewardOf(r) !== null)\n if (opts.minScore != null) {\n records = records.filter((r) => {\n const reward = rewardOf(r)\n return reward !== null && reward >= opts.minScore!\n })\n }\n\n const text = new Map(\n records.map((r) => [r.runId, { prompt: r.prompt!, completion: r.completion! }]),\n )\n const lookups = {\n promptOf: (id: string) => text.get(id)?.prompt ?? '',\n completionOf: (id: string) => text.get(id)?.completion ?? '',\n allowHeldOutTrainingData: opts.allowHeldOutTrainingData,\n }\n const { rows } = await mintRolloutRows(records, new InMemoryTraceStore())\n return buildRlDataset(rows, lookups, config)\n}\n","/**\n * Off-policy evaluation primitives.\n *\n * Standard inverse-probability-weighted (IPS), self-normalized\n * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the\n * value of a *target* policy given trajectories collected under a\n * *behavior* policy. This is the canonical RL eval task: \"we have last\n * week's runs, we changed the policy — how would the new one do without\n * re-running?\"\n *\n * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan\n * & Joachims 2015 for SNIPS) but the *application* to LLM-agent\n * evaluation needs care:\n *\n * - The \"policy\" is the (prompt, tool config, model snapshot) triple.\n * Two policies have the same probability over an action *iff* their\n * LLM call would emit the same token with the same probability —\n * which is generally unknowable without the model log-probs.\n * - For LLM agents, propensity scores must be supplied by the caller\n * (logged in the trace, recovered from token log-probs, or estimated\n * via a learned propensity model). We do NOT estimate propensity here.\n * - Doubly-robust requires two outputs from a Q-function: its prediction\n * for the logged action and its expectation under the target policy.\n * Consumers compute these with a tabular estimate, regression fit, or\n * learned reward model before constructing the trajectories.\n *\n * Bias / variance tradeoffs:\n * - IPS: unbiased; high variance for small overlap, infinite variance\n * when target has support outside behavior.\n * - SNIPS: lower variance, slight bias; usually preferred in practice.\n * - DR: doubly-robust — unbiased if either propensity OR Q-function is\n * correct. Lowest practical variance when Q is decent. Use this.\n *\n * Caveat the panel will land: on the LLM-agent setting, propensity scores\n * recovered from token log-probs are noisy, the action space is enormous,\n * and overlap is often poor. These estimators are useful but not magic;\n * complement with `replayCampaign` (exact replay where the request hashes\n * match) for high-confidence answers and OPE for the gap.\n */\n\nimport { ValidationError } from '../errors'\n\nexport interface OffPolicyTrajectory {\n /** Stable id, for traceability through the dataset. */\n runId: string\n /** Reward observed under the behavior policy (the realized outcome). */\n reward: number\n /**\n * Behavior-policy probability of the action that was taken. For LLM\n * agents this is typically `exp(sum(token_log_probs))` over the chosen\n * trajectory. Must be in (0, 1].\n */\n behaviorProb: number\n /**\n * Target-policy probability of the same action. For replay-style\n * counterfactual evaluation this is what the *new* policy would have\n * assigned to the *old* trajectory. Must be in [0, 1].\n */\n targetProb: number\n /**\n * Model-based reward prediction for the action selected by the behavior\n * policy: `Q_hat(context, loggedAction)`. Supply this together with\n * `vHatTarget` for contextual-bandit doubly-robust estimation.\n */\n qHatChosen?: number | null\n /**\n * Expected model-based reward under the target policy:\n * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.\n * Supply this together with `qHatChosen`. For an honest evaluation, both\n * values must come from a model cross-fitted or trained outside this row.\n */\n vHatTarget?: number | null\n}\n\nexport interface OffPolicyContributionCounts {\n /** Contributions using the contextual-bandit doubly-robust formula. */\n dr: number\n /** Contributions using exact IPS because no reward-model estimate was supplied. */\n ipsFallback: number\n}\n\nexport interface OffPolicyEstimate {\n /** Estimated value of the target policy. */\n value: number\n /** Standard error of the estimate. */\n standardError: number\n /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */\n effectiveSampleSize: number\n /** Number of trajectories used. */\n n: number\n /**\n * Diagnostic: maximum importance weight observed. Large values (>>10x\n * mean) are a red flag — variance is dominated by a few outliers.\n */\n maxImportanceWeight: number\n /** Populated by `doublyRobust` to expose which formula each row used. */\n contributionCounts?: OffPolicyContributionCounts\n}\n\nexport interface OffPolicyOptions {\n /**\n * Cap importance weights at this value (Ionides 2008 truncated IS) to\n * trade unbiasedness for variance reduction. Default `Infinity` (no cap).\n * Set e.g. `10` for stable estimates when the policies are close.\n */\n weightCap?: number\n /** Reward clipping range. Default `[0, 1]`. */\n rewardClip?: { low: number; high: number }\n}\n\n/**\n * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator\n * of E[reward under target policy]. Variance scales with the spread of\n * target/behavior ratios.\n */\nexport function inverseProbabilityWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n\n if (trajectories.length === 0) {\n return zeroEstimate()\n }\n\n const weights: number[] = []\n const weightedRewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `inverseProbabilityWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n weights.push(w)\n weightedRewards.push(w * r)\n if (w > maxW) maxW = w\n }\n const n = weights.length\n const value = weightedRewards.reduce((s, x) => s + x, 0) / n\n const variance = weightedRewards.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at\n * the cost of small bias (vanishing as N grows). The right default for\n * LLM-agent evaluation where overlap is often poor.\n */\nexport function selfNormalizedImportanceWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) return zeroEstimate()\n\n const weights: number[] = []\n const rewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `selfNormalizedImportanceWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n weights.push(w)\n rewards.push(clamp(t.reward, clip.low, clip.high))\n if (w > maxW) maxW = w\n }\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumWR = weights.reduce((s, w, i) => s + w * rewards[i]!, 0)\n const value = sumW === 0 ? 0 : sumWR / sumW\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n // Influence-function-based SE for SNIPS (Owen 2013, Ch. 9).\n const phi = weights.map((w, i) => w * (rewards[i]! - value))\n const variance = phi.reduce((s, x) => s + x * x, 0) / Math.max(1, sumW * sumW)\n return {\n value,\n standardError: Math.sqrt(variance),\n effectiveSampleSize: effN,\n n: trajectories.length,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).\n *\n * V_DR = (1/N) * sum_i [ v_hat_target_i\n * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]\n *\n * Unbiased if EITHER:\n * - the importance ratios are correct (IPS-style validity), OR\n * - the Q-hat function is correct (model-based validity).\n *\n * In practice both are imperfect, but the residual bias is the *product*\n * of both errors — much smaller than either alone. This is why DR is the\n * default in production OPE pipelines.\n *\n * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither\n * use the exact IPS contribution. `contributionCounts` makes the mix explicit\n * in the result.\n * Callers must cross-fit the Q-function or train it on independent rows;\n * fitting and evaluating Q on the same outcomes leaks the answer.\n */\nexport function doublyRobust(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) {\n return {\n ...zeroEstimate(),\n contributionCounts: { dr: 0, ipsFallback: 0 },\n }\n }\n\n const contributions: number[] = []\n const contributionCounts: OffPolicyContributionCounts = {\n dr: 0,\n ipsFallback: 0,\n }\n let maxW = 0\n let sumW = 0\n let sumW2 = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(`doublyRobust: behaviorProb must be > 0 (runId=${t.runId})`)\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n const rawQHatChosen = t.qHatChosen\n const rawVHatTarget = t.vHatTarget\n const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== undefined\n const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== undefined\n if (hasQHatChosen !== hasVHatTarget) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`,\n )\n }\n\n if (hasQHatChosen && hasVHatTarget) {\n if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`,\n )\n }\n const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high)\n const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high)\n contributions.push(vHatTarget + w * (r - qHatChosen))\n contributionCounts.dr += 1\n } else {\n contributions.push(w * r)\n contributionCounts.ipsFallback += 1\n }\n if (w > maxW) maxW = w\n sumW += w\n sumW2 += w * w\n }\n const n = contributions.length\n const value = contributions.reduce((s, x) => s + x, 0) / n\n const variance = contributions.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n contributionCounts,\n }\n}\n\n/**\n * Convenience: run all three estimators and return them side-by-side.\n * The recommended diagnostic — agreement across estimators is a much\n * stronger signal than any single one.\n */\nexport function offPolicyEstimateAll(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): { ips: OffPolicyEstimate; snips: OffPolicyEstimate; dr: OffPolicyEstimate } {\n return {\n ips: inverseProbabilityWeighting(trajectories, opts),\n snips: selfNormalizedImportanceWeighting(trajectories, opts),\n dr: doublyRobust(trajectories, opts),\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction zeroEstimate(): OffPolicyEstimate {\n return { value: 0, standardError: 0, effectiveSampleSize: 0, n: 0, maxImportanceWeight: 0 }\n}\n\nfunction clamp(x: number, lo: number, hi: number): number {\n if (!Number.isFinite(x)) return lo\n return Math.max(lo, Math.min(hi, x))\n}\n","/**\n * `PredictiveValidityResearcher` — concrete `Researcher` implementation\n * that drives selection from outcome-anchored predictive validity.\n *\n * Each method:\n *\n * - `inspectFailures(runs)` — synthesizes failure modes from the\n * bottom-quartile of `RunRecord`s on the configured proxy reward.\n * - `proposeChange(failures)` — proposes steering changes that target\n * the rubrics with the lowest predictive validity (decorative ones).\n * Either reduce their weight in the composite, or recalibrate them.\n * - `applyChange(changes, baseline)` — merges the proposed steering\n * into the experiment plan.\n * - `evaluateChange(plan)` — re-runs the predictive-validity check on\n * the post-change runs and reports the delta.\n *\n * The result is a closed loop: the rubric weights drift toward the ones\n * that actually predict deployment outcomes, automatically. Pair with\n * `runRLCampaign` for the full auto-research story.\n */\n\nimport type { GateDecision, SplitCoverage } from '../held-out-gate'\nimport type { OutcomeStore } from '../meta-eval/outcome-store'\nimport {\n type RubricPredictiveValidityReport,\n rubricPredictiveValidity,\n} from '../meta-eval/rubric-predictive-validity'\nimport type {\n ExperimentPlan,\n ExperimentResult,\n FailureMode,\n Researcher,\n SteeringChange,\n} from '../researcher'\nimport { type RunRecord, runTaskScore } from '../run-record'\n\nexport interface PredictiveValidityResearcherOptions {\n outcomes: OutcomeStore\n outcomeMetrics: string[]\n /** Score threshold below which a run counts as a \"failure.\" Default 0.5. */\n failureThreshold?: number\n /** Spearman bucket below which a rubric is \"decorative.\" Default 0.4. */\n decorativeThreshold?: number\n /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */\n steeringNamespace?: string\n /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */\n rubrics?: string[]\n /**\n * Snapshot stash hook — called with the most recent predictive-validity\n * report. Useful when a downstream system wants to log rubric drift over\n * time. Default no-op.\n */\n onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>\n}\n\n/**\n * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:\n * rubrics that don't predict deployment outcomes don't earn weight.\n */\nexport class PredictiveValidityResearcher implements Researcher {\n private opts: PredictiveValidityResearcherOptions\n private lastReport: RubricPredictiveValidityReport | null = null\n\n constructor(opts: PredictiveValidityResearcherOptions) {\n this.opts = opts\n }\n\n async inspectFailures(runs: RunRecord[]): Promise<FailureMode[]> {\n const threshold = this.opts.failureThreshold ?? 0.5\n const failures: FailureMode[] = []\n // Ungated: the researcher reports what the runs actually scored. A gamed\n // run scored high and is therefore NOT a low-score failure mode — calling it\n // one here would attribute the wrong failure to the candidate.\n const failingRuns = runs.filter((r) => {\n const score = runTaskScore(r)\n return typeof score === 'number' && score < threshold\n })\n if (failingRuns.length === 0) return failures\n\n // Group failures by candidateId — the researcher's primary handle is\n // \"this candidate is producing low-scoring outputs in this scenario.\"\n const grouped = new Map<string, RunRecord[]>()\n for (const r of failingRuns) {\n const arr = grouped.get(r.candidateId) ?? []\n arr.push(r)\n grouped.set(r.candidateId, arr)\n }\n\n for (const [candidateId, group] of grouped.entries()) {\n const meanScore =\n group.reduce((s, r) => {\n const score = runTaskScore(r)\n if (score === undefined) {\n throw new Error(`failing run ${r.runId} unexpectedly has no task score`)\n }\n return s + score\n }, 0) / group.length\n failures.push({\n code: `low-score-${candidateId}`,\n description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,\n evidence: {\n runIds: group.slice(0, 8).map((r) => r.runId),\n samples: group.length,\n },\n })\n }\n return failures\n }\n\n async proposeChange(failures: FailureMode[]): Promise<SteeringChange[]> {\n if (failures.length === 0) return []\n\n // Without a prior report, return a single \"collect more outcome data\"\n // change — the researcher refuses to reweight rubrics from zero evidence.\n if (this.lastReport === null) {\n return [\n {\n kind: 'threshold',\n payload: { directive: 'researcher.collect-more-outcomes' },\n rationale:\n 'predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists',\n },\n ]\n }\n\n const decorativeThreshold = this.opts.decorativeThreshold ?? 0.4\n const changes: SteeringChange[] = []\n\n for (const ranking of this.lastReport.ranked) {\n if (ranking.verdict === 'load_bearing') continue\n if (Math.abs(ranking.spearman) >= decorativeThreshold) continue\n changes.push({\n kind: 'reviewer_prompt',\n payload: {\n rubric: ranking.rubric,\n action: 'down-weight',\n spearman: ranking.spearman,\n bestOutcome: ranking.bestOutcome,\n },\n rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,\n expectedDelta: -Math.max(0, 0.05 - Math.abs(ranking.spearman)),\n })\n }\n for (const ranking of this.lastReport.ranked.slice(0, 1)) {\n if (ranking.verdict !== 'load_bearing') continue\n changes.push({\n kind: 'reviewer_prompt',\n payload: {\n rubric: ranking.rubric,\n action: 'up-weight',\n spearman: ranking.spearman,\n bestOutcome: ranking.bestOutcome,\n },\n rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,\n expectedDelta: Math.max(0, Math.abs(ranking.spearman) - 0.5) * 0.1,\n })\n }\n return changes\n }\n\n async applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan> {\n // Merge proposed changes into the plan's `changes` array, preserving\n // any changes the baseline already had.\n return {\n ...baseline,\n changes: [...baseline.changes, ...changes],\n }\n }\n\n async evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult> {\n // The researcher contract takes a *plan* and returns a *result* —\n // implementations that only understand re-scoring runs can produce a\n // \"no-op\" gate decision and let the caller drive the actual sweep.\n // Real evaluators (CallbackResearcher) execute the plan; we report.\n const emptyGate: GateDecision = {\n promote: false,\n candidateId: plan.proposedCandidateId,\n baselineId: plan.baselineCandidateId,\n evidence: {\n productiveRuns: 0,\n unpairedCandidateRuns: 0,\n unpairedBaselineRuns: 0,\n medianPairedDelta: null,\n deltaStatistic: 'median_bootstrap',\n decidingDelta: null,\n pairedCI: null,\n pairedPValue: null,\n mcnemar: null,\n binaryScale: null,\n tieFraction: null,\n searchScore: null,\n holdoutScore: null,\n overfitGap: null,\n baselineOverfitGap: null,\n medianCandidateCost: null,\n medianBaselineCost: null,\n realnessGatedRuns: 0,\n // Nothing was dealt, so nothing was answered — this researcher never\n // runs the sweep, it only reports that the caller must.\n holdoutCoverage: emptyCoverage(),\n searchCoverage: emptyCoverage(),\n },\n reason:\n 'predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].',\n rejectionCode: 'few_runs',\n }\n return {\n plan,\n runs: [],\n gateDecision: emptyGate,\n }\n }\n\n /**\n * Run the predictive-validity check explicitly against a fresh RunRecord\n * set. Updates the researcher's cached report so subsequent\n * `proposeChange` calls have evidence to draw from.\n */\n async runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport> {\n const report = await rubricPredictiveValidity({\n runs,\n outcomes: this.opts.outcomes,\n outcomeMetrics: this.opts.outcomeMetrics,\n rubrics: this.opts.rubrics,\n })\n if (this.opts.onReport) await this.opts.onReport(report)\n this.lastReport = report\n return report\n }\n\n /**\n * Force-feed a predictive-validity report into the researcher state —\n * useful when the consumer ran the report out-of-band and wants the\n * researcher's later proposals informed by it.\n */\n setReport(report: RubricPredictiveValidityReport): void {\n this.lastReport = report\n }\n\n getLastReport(): RubricPredictiveValidityReport | null {\n return this.lastReport\n }\n}\n\n/** Coverage of a split that was never dealt any work. */\nfunction emptyCoverage(): SplitCoverage {\n return { dealt: 0, answered: 0, unscoredPairs: 0, candidateOnly: 0, baselineOnly: 0, coverage: 0 }\n}\n","/**\n * Preference dataset extraction from canonical minted rollout lines.\n *\n * Production RLHF / DPO / KTO / SimPO pipelines need preference triples:\n * `(prompt, chosen, rejected)`. The campaign artifact already contains the\n * ingredients — every (variantId, scenarioId, seed) cell is a candidate\n * that ran the same prompt against the same scenario, scored by the same\n * judge — but turning that into a clean preference dataset requires\n * deciding *what counts as a preference*.\n *\n * This module ships three preference-extraction strategies with explicit\n * tradeoffs, plus a unified output type compatible with HuggingFace TRL,\n * Anthropic finetuning JSONL, and OpenAI fine-tuning APIs. The strategies\n * are deliberately not auto-magical — picking the wrong one corrupts the\n * gradient.\n *\n * Strategies:\n *\n * 1. **`paired-by-scenario-and-seed`** — exact-match comparisons. For\n * each scenario × seed pair, compare every (variantA, variantB) on\n * that exact (scenario, seed). Matches scenarios so the comparison\n * isolates variant effects. Highest signal-to-noise; smallest\n * dataset (only matched pairs count).\n *\n * 2. **`paired-by-scenario`** — looser matching. For each scenario,\n * compare every (variantA, variantB) where both have ≥ 1 run on the\n * same scenario. Aggregates across seeds to compute mean scores per\n * (variant, scenario), then forms preferences from the means. More\n * data, lower per-pair signal.\n *\n * 3. **`top-vs-bottom`** — coarsest. Within each scenario, the highest-\n * scoring run is `chosen`, the lowest is `rejected`. Smallest dataset\n * per scenario but biggest score gap per pair. Useful for early\n * bootstrapping when you have few variants.\n *\n * The output `PreferenceTriple` is *agent-eval-canonical* but trivially\n * mappable to TRL's `DPODataset` shape (`prompt`, `chosen`, `rejected`)\n * via the `toTRLFormat` helper, which resolves real prompt/completion text\n * through the same lookups `toDpoRows` takes (`./exporters` carries the\n * richer row with margin + metadata).\n *\n * Input discipline: the function accepts only `MintedRolloutLine[]`, whose\n * reward and authenticity fields have already been validated. `search` is the\n * default split; held-out pairing requires an explicit opt-in, while `dev` and\n * `canary` remain evaluation-only.\n */\n\nimport type { MintedRolloutLine, RolloutSplit } from '../rollout/schema'\nimport type { DpoLookups } from './exporters'\nimport {\n admitUngatedByInvocation,\n type LineContextRequirement,\n type RolloutLineContext,\n trainableLineReward,\n} from './rollout-input'\n\nexport type PreferenceStrategy =\n | 'paired-by-scenario-and-seed'\n | 'paired-by-scenario'\n | 'top-vs-bottom'\n\nexport interface PreferenceTriple {\n /** The scenario (input) the variants were run against. */\n scenarioId: string\n /** RunRecord ids on each side, for traceability. */\n chosenRunId: string\n rejectedRunId: string\n /** Variant ids — load-bearing for the RL update. */\n chosenVariantId: string\n rejectedVariantId: string\n /** The score gap between chosen and rejected. Larger = stronger signal. */\n marginScore: number\n /**\n * Optional `(chosen_score, rejected_score)` pair for soft-margin DPO\n * variants. Omitted for `top-vs-bottom` runs that don't carry meaningful\n * scalar gaps.\n */\n scores?: { chosen: number; rejected: number }\n /** Tie-breaker — when multiple seeds match this scenario, the one used. */\n seed?: number\n /**\n * Free-form metadata propagated from the rollout lines, such as original\n * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.\n */\n meta: {\n chosenPromptHash: string\n rejectedPromptHash: string\n chosenConfigHash: string\n rejectedConfigHash: string\n chosenModel: string\n rejectedModel: string\n }\n}\n\nexport interface ExtractPreferencesOptions {\n strategy?: PreferenceStrategy\n /**\n * Minimum score gap required to admit a pair. Pairs below this are\n * dropped — they're noise, not signal. Default 0.05 (5% of [0,1]).\n */\n minMargin?: number\n /**\n * Optional split filter. Without one, only search is included.\n * Holdout requires `allowHeldOutTrainingData: true`; dev and canary are\n * evaluation-only.\n */\n split?: RolloutSplit\n /** Named opt-in required before held-out lines may be paired. */\n allowHeldOutTrainingData?: boolean\n}\n\nexport interface PreferenceExtractionReport {\n pairs: PreferenceTriple[]\n /** Number of (scenario, seed) cells inspected. */\n cellsInspected: number\n /** Number of pairs filtered by `minMargin`. */\n pairsBelowMargin: number\n /** Number of cells with only one variant (no comparison possible). */\n cellsSingleton: number\n /** Strategy used. */\n strategy: PreferenceStrategy\n /**\n * Lines dropped before pairing because they carry no `candidate_id`. A\n * preference is a statement about two candidates, so a line that names none\n * cannot be paired.\n */\n linesWithoutCandidateId: number\n}\n\n/** The split each path pairs by default: training data comes from search. */\nconst SPLIT_DEFAULT: RolloutSplit = 'search'\n\n/**\n * The only shape the pairing strategies see.\n */\ninterface PairingCandidate {\n scenarioId: string\n runId: string\n candidateId: string\n /** null when a line records no seed. */\n seed: number | null\n score: number\n promptHash: string\n configHash: string\n model: string\n}\n\n/**\n * Convert rollout lines to preference triples for RL training.\n *\n * Returns a structured report so callers can see how much data was\n * dropped and why (low-margin pairs, singleton cells). For production\n * pipelines, you usually want to:\n *\n * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds\n * 2. Mint the runs with `mintRolloutRows` and call this with\n * `strategy: 'paired-by-scenario-and-seed'`\n * 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with\n * prompt/completion resolvers and pipe to your DPO trainer\n *\n * The gate is what makes a preference dataset safe: ordered on an ungated\n * score, a gamed run with an inflated number becomes the `chosen` side and DPO\n * is trained to prefer the gaming trajectory over its honest sibling. A gated\n * line arrives here already scored 0, so it sinks to `rejected`.\n */\nexport function extractPreferences(\n lines: MintedRolloutLine[],\n opts: ExtractPreferencesOptions = {},\n): PreferenceExtractionReport {\n const strategy = opts.strategy ?? 'paired-by-scenario-and-seed'\n const minMargin = opts.minMargin ?? 0.05\n const requestedSplit = opts.split\n if (requestedSplit === 'holdout' && opts.allowHeldOutTrainingData !== true) {\n throw new Error('extractPreferences: split \"holdout\" requires allowHeldOutTrainingData: true')\n }\n if (requestedSplit === 'dev' || requestedSplit === 'canary') {\n throw new Error(\n `extractPreferences: split \"${requestedSplit}\" is evaluation-only; train from \"search\"`,\n )\n }\n const candidates = candidatesFromLines(lines, opts)\n const report = pairCandidates(candidates.rows, strategy, minMargin)\n return { ...report, linesWithoutCandidateId: candidates.withoutCandidateId }\n}\n\ninterface NormalizedInput {\n rows: PairingCandidate[]\n withoutCandidateId: number\n}\n\nfunction candidatesFromLines(\n lines: MintedRolloutLine[],\n opts: ExtractPreferencesOptions,\n): NormalizedInput {\n const split = opts.split ?? SPLIT_DEFAULT\n const rows: PairingCandidate[] = []\n let withoutCandidateId = 0\n for (const line of lines) {\n if (line.task.split !== split) continue\n if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {\n continue\n }\n const score = trainableLineReward(line)\n if (score === null) continue\n const candidateId = line.candidate_id\n if (candidateId === null || candidateId === undefined || candidateId.length === 0) {\n withoutCandidateId++\n continue\n }\n rows.push({\n scenarioId: line.task.instance_id,\n runId: line.run_id,\n candidateId,\n seed: line.task.seed,\n score,\n // `policy.*` is nullable on the wire; a minted line always carries these\n // (RunRecord makes them mandatory). Empty string marks \"not recorded\" so\n // `toTRLFormat`'s hash lookup fails visibly instead of silently matching.\n promptHash: line.policy.prompt_hash ?? '',\n configHash: line.policy.config_hash ?? '',\n model: line.policy.model ?? '',\n })\n }\n return { rows, withoutCandidateId }\n}\n\nfunction pairCandidates(\n scoredEntries: PairingCandidate[],\n strategy: PreferenceStrategy,\n minMargin: number,\n): Omit<PreferenceExtractionReport, 'linesWithoutCandidateId'> {\n const pairs: PreferenceTriple[] = []\n let pairsBelowMargin = 0\n let cellsSingleton = 0\n let cellsInspected = 0\n\n if (strategy === 'paired-by-scenario-and-seed') {\n // Group by the canonical (scenarioId, seed) identity.\n const groups = new Map<string, PairingCandidate[]>()\n for (const e of scoredEntries) {\n const key = `${e.scenarioId}::${e.seed}`\n const arr = groups.get(key) ?? []\n arr.push(e)\n groups.set(key, arr)\n }\n\n for (const members of groups.values()) {\n cellsInspected++\n if (members.length < 2) {\n cellsSingleton++\n continue\n }\n for (let i = 0; i < members.length; i++) {\n for (let j = i + 1; j < members.length; j++) {\n const a = members[i]!\n const b = members[j]!\n if (a.candidateId === b.candidateId) continue\n const result = makePair(a, b, a.scenarioId, minMargin)\n if (result.kind === 'admit') pairs.push(result.pair)\n else pairsBelowMargin++\n }\n }\n }\n } else if (strategy === 'paired-by-scenario') {\n // Group by scenarioId → average per (variantId, scenarioId) across seeds.\n const byScenarioVariant = new Map<\n string,\n Map<string, { entry: PairingCandidate; sum: number; n: number }>\n >()\n for (const e of scoredEntries) {\n let perScenario = byScenarioVariant.get(e.scenarioId)\n if (!perScenario) {\n perScenario = new Map()\n byScenarioVariant.set(e.scenarioId, perScenario)\n }\n const cur = perScenario.get(e.candidateId)\n if (cur) {\n cur.sum += e.score\n cur.n++\n } else perScenario.set(e.candidateId, { entry: e, sum: e.score, n: 1 })\n }\n for (const [sid, perVariant] of byScenarioVariant.entries()) {\n cellsInspected++\n const arr = [...perVariant.values()].map((agg) => ({\n ...agg.entry,\n score: agg.sum / agg.n,\n }))\n if (arr.length < 2) {\n cellsSingleton++\n continue\n }\n for (let i = 0; i < arr.length; i++) {\n for (let j = i + 1; j < arr.length; j++) {\n const result = makePair(arr[i]!, arr[j]!, sid, minMargin)\n if (result.kind === 'admit') pairs.push(result.pair)\n else pairsBelowMargin++\n }\n }\n }\n } else {\n // top-vs-bottom: per scenario, top vs bottom only.\n const byScenario = new Map<string, PairingCandidate[]>()\n for (const e of scoredEntries) {\n const arr = byScenario.get(e.scenarioId) ?? []\n arr.push(e)\n byScenario.set(e.scenarioId, arr)\n }\n for (const [sid, arr] of byScenario.entries()) {\n cellsInspected++\n if (arr.length < 2) {\n cellsSingleton++\n continue\n }\n const sorted = [...arr].sort((a, b) => a.score - b.score)\n const top = sorted[sorted.length - 1]!\n const bot = sorted[0]!\n if (top.candidateId === bot.candidateId) {\n cellsSingleton++\n continue\n }\n const result = makePair(bot, top, sid, minMargin)\n if (result.kind === 'admit') pairs.push(result.pair)\n else pairsBelowMargin++\n }\n }\n\n return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy }\n}\n\nconst PREFERENCE_RUN_IDS = (t: PreferenceTriple): readonly string[] => [\n t.chosenRunId,\n t.rejectedRunId,\n]\n\nconst TRL_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'TRL preference export',\n contextType: 'RolloutLineContext',\n because:\n 'a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair.',\n}\n\nconst ANTHROPIC_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'Anthropic preference export',\n contextType: 'RolloutLineContext',\n because:\n 'a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one.',\n}\n\n/**\n * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`\n * where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes\n * would optimize the policy toward emitting hex digests. Neither the prompt\n * nor the completions live on the triple (it carries only run ids and hashes),\n * so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`\n * takes, keyed by run id, and this function resolves real text.\n *\n * The chosen and rejected sides of a valid pair share one prompt; resolving\n * both and comparing catches lookup bugs (a stale map keyed by the wrong id)\n * before they ship a row whose prompt does not match its rejected completion.\n *\n * `context` is REQUIRED: this is the third exporter over the identical\n * line-less input class, and the round that hardened `toPrmRows` while leaving\n * `toDpoRows` open is why every one of them now takes the same argument and\n * runs the same admission rule.\n */\nexport async function toTRLFormat(\n triples: PreferenceTriple[],\n lookups: DpoLookups,\n context: RolloutLineContext,\n): Promise<Array<{ prompt: string; chosen: string; rejected: string }>> {\n const admitted = admitUngatedByInvocation(\n triples,\n PREFERENCE_RUN_IDS,\n context,\n TRL_CONTEXT_REQUIREMENT,\n )\n const out: Array<{ prompt: string; chosen: string; rejected: string }> = []\n for (const t of admitted) {\n const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([\n Promise.resolve(lookups.promptOf(t.chosenRunId)),\n Promise.resolve(lookups.promptOf(t.rejectedRunId)),\n Promise.resolve(lookups.completionOf(t.chosenRunId)),\n Promise.resolve(lookups.completionOf(t.rejectedRunId)),\n ])\n if (chosenPrompt !== rejectedPrompt) {\n throw new Error(\n `toTRLFormat: preference \"${t.chosenRunId}\"/\"${t.rejectedRunId}\" resolves to different prompts`,\n )\n }\n out.push({ prompt: chosenPrompt, chosen, rejected })\n }\n return out\n}\n\n/**\n * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`\n * shape. Same caveat as TRL: prompt + outputs are content the caller has\n * to map back from the run record / raw event log.\n *\n * `context` is REQUIRED — see `toTRLFormat`. The emitted `margin` is a number\n * derived from the two runs' rewards, so this row is training signal even\n * though it ships no completion text.\n */\nexport function toAnthropicFormat(\n triples: PreferenceTriple[],\n context: RolloutLineContext,\n): Array<{ scenarioId: string; chosenRunId: string; rejectedRunId: string; margin: number }> {\n return admitUngatedByInvocation(\n triples,\n PREFERENCE_RUN_IDS,\n context,\n ANTHROPIC_CONTEXT_REQUIREMENT,\n ).map((t) => ({\n scenarioId: t.scenarioId,\n chosenRunId: t.chosenRunId,\n rejectedRunId: t.rejectedRunId,\n margin: t.marginScore,\n }))\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction makePair(\n a: PairingCandidate,\n b: PairingCandidate,\n scenarioId: string,\n minMargin: number,\n): { kind: 'admit'; pair: PreferenceTriple } | { kind: 'reject' } {\n const margin = Math.abs(a.score - b.score)\n if (margin < minMargin) return { kind: 'reject' }\n const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a]\n const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : undefined\n return {\n kind: 'admit',\n pair: {\n scenarioId,\n chosenRunId: chosen.runId,\n rejectedRunId: rejected.runId,\n chosenVariantId: chosen.candidateId,\n rejectedVariantId: rejected.candidateId,\n marginScore: chosen.score - rejected.score,\n scores: { chosen: chosen.score, rejected: rejected.score },\n seed,\n meta: {\n chosenPromptHash: chosen.promptHash,\n rejectedPromptHash: rejected.promptHash,\n chosenConfigHash: chosen.configHash,\n rejectedConfigHash: rejected.configHash,\n chosenModel: chosen.model,\n rejectedModel: rejected.model,\n },\n },\n }\n}\n","/**\n * Process reward extraction — step-level credit assignment from trace spans.\n *\n * RL on long-horizon agents needs *step-level* rewards, not run-level\n * ones. The classic credit-assignment problem (Sutton & Barto) requires\n * knowing which sub-decisions in a trajectory contributed to the\n * outcome. Modern systems (DeepSeek-R1, OpenAI o-series, Lightman et al.\n * \"Let's Verify Step by Step\" 2023) train *process reward models* (PRMs)\n * that score every step, then do RL with the PRM as the reward signal.\n *\n * This module extracts `StepReward[]` from trace spans — one per\n * meaningful step — and ships:\n *\n * 1. `extractStepRewards(store, runId, opts)` — span → step-reward\n * conversion using configurable per-span scorers (LLM judge over the\n * span output, deterministic checkers, or a learned PRM).\n * 2. `runwiseStepRewardSummary(stepRewards)` — aggregate the per-step\n * signal into a credit-assignment-aware run-level score.\n * 3. `prmTrainingPairs(stepRewards, options)` — produce the\n * `(prefix, suffix_chosen, suffix_rejected)` triples that PRM\n * training pipelines consume.\n *\n * What we ship: the *extraction* and *aggregation* infrastructure plus\n * the data shape PRM training expects. We do NOT ship the actual PRM\n * training (gradient descent over a transformer is out of scope for a\n * TS package). The interface is the contract; downstream consumers wire\n * their preferred trainer.\n *\n * Caveat the panel will land: this is descriptive credit assignment\n * (which steps correlate with outcome), not causal credit assignment\n * (which steps caused outcome). For causal claims you need\n * counterfactual rollouts or a learned dynamics model. Future work; the\n * descriptive version is what production PRM training actually uses.\n */\n\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface StepReward {\n /** Trace span this reward attaches to. */\n spanId: string\n runId: string\n /** Index in the trajectory (0-based, in started-at order). */\n stepIndex: number\n /** Span kind (typically 'tool', 'llm', 'judge'). */\n kind: Span['kind']\n /** Span name — for the consumer's downstream filtering. */\n name: string\n /** Step-level reward in [0, 1]. */\n reward: number\n /**\n * Determinism class. Mirrors the verifiable-reward distinction:\n * deterministic = test/compile/schema check; probabilistic = LLM judge.\n */\n determinism: 'deterministic' | 'probabilistic'\n /** Optional rationale / evidence — the trainer typically discards. */\n rationale?: string\n /** Optional weight — how much this step contributes to credit assignment. */\n weight?: number\n}\n\nexport interface StepScorer {\n /** Span kinds this scorer applies to. */\n appliesTo: Span['kind'][]\n /** Returns null to skip the span; returns a `StepReward` shape (without index/runId/spanId, which are filled in). */\n score(span: Span): Promise<Omit<StepReward, 'spanId' | 'runId' | 'stepIndex'>> | null | undefined\n}\n\nexport interface ExtractStepRewardsOptions {\n /**\n * Ordered list of scorers. Each span runs through scorers in order;\n * the first non-null result wins. If no scorer applies, the span is\n * skipped (not all spans are training-worthy).\n */\n scorers: StepScorer[]\n /** Optional filter — return null to drop the span entirely before scoring. */\n preFilter?: (span: Span) => boolean\n}\n\nexport async function extractStepRewards(\n store: TraceStore,\n runId: string,\n opts: ExtractStepRewardsOptions,\n): Promise<StepReward[]> {\n const spans = await store.spans({ runId })\n const ordered = [...spans].sort((a, b) => a.startedAt - b.startedAt)\n const out: StepReward[] = []\n let idx = 0\n for (const span of ordered) {\n if (opts.preFilter && !opts.preFilter(span)) continue\n let scored: Awaited<ReturnType<StepScorer['score']>> = null\n for (const s of opts.scorers) {\n if (!s.appliesTo.includes(span.kind)) continue\n const r = await s.score(span)\n if (r) {\n scored = r\n break\n }\n }\n if (!scored) continue\n out.push({\n spanId: span.spanId,\n runId,\n stepIndex: idx++,\n kind: span.kind,\n name: span.name,\n reward: scored.reward,\n determinism: scored.determinism,\n rationale: scored.rationale,\n weight: scored.weight,\n })\n }\n return out\n}\n\nexport interface RunwiseStepSummary {\n runId: string\n totalSteps: number\n meanReward: number\n /** Sum-of-rewards (weighted by `weight ?? 1`). Use as the run-level proxy. */\n sumWeightedReward: number\n /** Fraction of steps where reward < 0.5 — proxy for \"where the policy was wrong.\" */\n failureFraction: number\n /** Maximum drop in reward between consecutive steps — diagnoses a step where things went sideways. */\n worstStepDelta: number\n worstStepIndex: number | null\n}\n\nexport function runwiseStepRewardSummary(stepRewards: StepReward[]): RunwiseStepSummary {\n if (stepRewards.length === 0) {\n return {\n runId: '',\n totalSteps: 0,\n meanReward: 0,\n sumWeightedReward: 0,\n failureFraction: 0,\n worstStepDelta: 0,\n worstStepIndex: null,\n }\n }\n const runId = stepRewards[0]!.runId\n let sumW = 0\n let sumWR = 0\n let failures = 0\n let worstDelta = 0\n let worstIdx: number | null = null\n let prev = stepRewards[0]!.reward\n for (let i = 0; i < stepRewards.length; i++) {\n const s = stepRewards[i]!\n const w = s.weight ?? 1\n sumW += w\n sumWR += w * s.reward\n if (s.reward < 0.5) failures++\n if (i > 0) {\n const delta = s.reward - prev\n if (delta < worstDelta) {\n worstDelta = delta\n worstIdx = i\n }\n prev = s.reward\n } else {\n prev = s.reward\n }\n }\n return {\n runId,\n totalSteps: stepRewards.length,\n meanReward: sumW === 0 ? 0 : sumWR / sumW,\n sumWeightedReward: sumWR,\n failureFraction: failures / stepRewards.length,\n worstStepDelta: worstDelta,\n worstStepIndex: worstIdx,\n }\n}\n\nexport interface PrmTrainingTriple {\n /** Prefix run-id (or composite key) — the trajectory up to step k-1. */\n prefixRunId: string\n prefixStepIndex: number\n /** The step that came next on a high-reward trajectory. */\n chosenSpanId: string\n chosenReward: number\n /** A step from a divergent low-reward trajectory at the same prefix length. */\n rejectedSpanId: string\n rejectedReward: number\n /** The prefix run came from this run; the rejected step came from `rejectedRunId`. */\n rejectedRunId: string\n marginScore: number\n}\n\n/**\n * Build PRM training triples. The shape: pair runs that share an early\n * prefix (same scenario, same first N steps) and diverge later — at the\n * point of divergence, the high-reward run's next step is `chosen`, the\n * low-reward run's next step is `rejected`. This is the canonical PRM\n * training data shape from Lightman et al. and DeepSeek-R1 process\n * supervision.\n *\n * Implementation note: we don't have a way to detect \"same prefix\" in\n * the general agent setting (token-level prefixes require hashing model\n * outputs). The current heuristic groups by `(scenarioId, prefixSpanName\n * sequence)` — runs are paired when their first K span names match. For\n * production use this should be replaced with a proper trajectory-prefix\n * hash; the heuristic is good enough for early-stage scaffolding.\n */\nexport function prmTrainingPairs(\n stepRewardsByRun: Map<string, StepReward[]>,\n opts: { minMargin?: number; minPrefixLength?: number } = {},\n): PrmTrainingTriple[] {\n const minMargin = opts.minMargin ?? 0.2\n const minPrefix = opts.minPrefixLength ?? 1\n const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({ runId, steps }))\n const triples: PrmTrainingTriple[] = []\n\n for (let i = 0; i < runs.length; i++) {\n for (let j = i + 1; j < runs.length; j++) {\n const a = runs[i]!\n const b = runs[j]!\n const minLen = Math.min(a.steps.length, b.steps.length)\n if (minLen < minPrefix + 1) continue\n\n // Find the first index where the trajectories diverge: either by\n // step structure (kind/name mismatch) OR by reward gap ≥ minMargin.\n // Names that match but rewards that differ ARE divergence — that's\n // the canonical PRM training case (same step structure, different\n // outcomes via state/context).\n let divergenceIdx = -1\n for (let k = 0; k < minLen; k++) {\n const sa = a.steps[k]!\n const sb = b.steps[k]!\n const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name\n const rewardGap = Math.abs(sa.reward - sb.reward)\n if (structuralDivergence || rewardGap >= minMargin) {\n divergenceIdx = k\n break\n }\n }\n if (divergenceIdx < 0) continue\n if (divergenceIdx < minPrefix) continue\n\n const aNext = a.steps[divergenceIdx]!\n const bNext = b.steps[divergenceIdx]!\n const margin = Math.abs(aNext.reward - bNext.reward)\n if (margin < minMargin) continue\n\n const chosen = aNext.reward > bNext.reward ? aNext : bNext\n const rejected = aNext.reward > bNext.reward ? bNext : aNext\n const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId\n const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId\n triples.push({\n prefixRunId: chosenRun,\n prefixStepIndex: divergenceIdx - 1,\n chosenSpanId: chosen.spanId,\n chosenReward: chosen.reward,\n rejectedSpanId: rejected.spanId,\n rejectedReward: rejected.reward,\n rejectedRunId: rejectedRun,\n marginScore: chosen.reward - rejected.reward,\n })\n }\n }\n return triples\n}\n","/**\n * `runRLCampaign` — top-level orchestrator that runs the matrix and\n * produces every RL-ready artifact in one call.\n *\n * Wires:\n * 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)\n * 2. `extractVerifiableRewardsFromRecords` over the runs, separating deterministic\n * from probabilistic reward sources for the trainer\n * 3. `extractPreferences` to produce DPO/PPO/KTO triples\n * 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)\n * 5. `rubricPredictiveValidity` against an outcome store, when provided\n * 6. `detectRewardHacking` as a standing hygiene check\n * 7. Trainer-format export rows ready for prime-rl / TRL / verl\n *\n * The output `RLCampaignResult` is a single, audit-ready artifact: every\n * stage's output is in there. The consumer's downstream fits in a single\n * line: pass `result.preferences.pairs` to a DPO trainer,\n * `result.trainerRows.grpo` to GRPO, or `result.campaign.runs` plus\n * `result.rewardSignals` to a custom RL loop.\n */\n\nimport {\n type EvalCampaignOptions,\n type EvalCampaignResult,\n type FailedRun,\n runEvalCampaign,\n} from '../eval-campaign'\nimport type { OutcomeStore } from '../meta-eval/outcome-store'\nimport {\n type RubricPredictiveValidityReport,\n rubricPredictiveValidity,\n} from '../meta-eval/rubric-predictive-validity'\nimport { mintRolloutRows } from '../rollout/mint'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport { evaluateInterimReleaseConfidence, type InterimReleaseConfidence } from '../sequential'\nimport { InMemoryTraceStore } from '../trace/store'\nimport {\n type DpoExportRow,\n type DpoLookups,\n type GrpoExportRow,\n type GrpoLookups,\n type SftExportRow,\n type SftLookups,\n toDpoRows,\n toGrpoRows,\n toSftRows,\n} from './exporters'\nimport {\n type ExtractPreferencesOptions,\n extractPreferences,\n type PreferenceExtractionReport,\n} from './preferences'\nimport { detectRewardHacking, type RewardHackingReport } from './reward-hacking'\nimport {\n extractVerifiableRewardsFromRecords,\n type VerifiableReward,\n type VerifiableRewardExtractionOptions,\n} from './verifiable-reward'\n\nexport interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {\n /** Preference-extraction options. Default uses paired-by-scenario-and-seed with min-margin 0.05. */\n preferences?: ExtractPreferencesOptions\n /** Verifiable-reward extraction options. */\n verifiableReward?: VerifiableRewardExtractionOptions\n /** Outcome store + metric names — when supplied, runs `rubricPredictiveValidity` post-campaign. */\n outcomeStore?: OutcomeStore\n outcomeMetrics?: string[]\n /** Anytime-valid sequential evaluation options. */\n sequential?: {\n alpha?: number\n bound?: number\n rope?: { low: number; high: number }\n /**\n * Smallest acceptable `answered / dealt` fraction of paired cells, required\n * of EVERY candidate before the interim verdict is computed at all. Default\n * 1: every cell the comparison was dealt must carry a score on both arms.\n *\n * The default is 1 because `recommendation.decision` can be `promote_now`,\n * and a recommendation computed over \"the cells that happened to pair\" is\n * computed over a set the candidate selected by failing. Below this,\n * `interimConfidence` is null and `deltaCoverage` says by how much — never a\n * silent 0. Must be in [0, 1]; anything else throws rather than clamping.\n */\n minDeltaCoverage?: number\n }\n /** Trainer-format export lookups. When provided, the orchestrator builds the corresponding rows. */\n trainerExport?: {\n dpo?: DpoLookups\n grpo?: GrpoLookups\n sft?: SftLookups\n }\n}\n\n/**\n * How much of one candidate's DEALT paired work produced a usable delta.\n *\n * `answered + unscoredCandidate + unscoredComparator + unmatched === dealt` — a\n * complete, mutually exclusive partition of the (scenarioId, seed) cells the\n * comparison was dealt, so no cell can leave the delta series without appearing\n * in exactly one bucket.\n */\nexport interface PairedDeltaCoverage {\n candidateId: string\n /** Cells the comparison was dealt: a run on either arm. */\n dealt: number\n /** Dealt cells scored on BOTH arms — the deltas the verdict is read from. */\n answered: number\n /** Dealt cells this candidate ran and produced no usable score for. */\n unscoredCandidate: number\n /** Dealt cells the comparator ran and produced no usable score for. */\n unscoredComparator: number\n /** Dealt cells only one of the two arms produced a run for. */\n unmatched: number\n /** `answered / dealt`, or 0 when nothing was dealt. */\n coverage: number\n}\n\nexport interface RLCampaignResult {\n campaign: EvalCampaignResult\n /** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */\n rewardSignals: Array<{ runId: string; reward: VerifiableReward | null }>\n /** Preference extraction report. */\n preferences: PreferenceExtractionReport\n /** Anytime-valid interim verdict over the paired deltas (vs comparator).\n * Null when no comparator was configured, when nothing paired, or when a\n * candidate fell below `sequential.minDeltaCoverage` — read `deltaCoverage`\n * to tell those apart. */\n interimConfidence: InterimReleaseConfidence | null\n /** Answered / dealt paired cells per candidate — the denominator behind\n * `interimConfidence`, reported on EVERY path including the ones where the\n * verdict was refused. Empty when no comparator was configured. */\n deltaCoverage: PairedDeltaCoverage[]\n /** Standing reward-hacking hygiene check. */\n rewardHacking: RewardHackingReport\n /** Predictive validity, when an outcome store was supplied. */\n predictiveValidity: RubricPredictiveValidityReport | null\n /** Trainer-export rows, populated only for the formats the caller requested via `trainerExport`. */\n trainerRows: {\n dpo?: DpoExportRow[]\n grpo?: GrpoExportRow[]\n sft?: SftExportRow[]\n }\n /**\n * One-line top-level summary the consumer can log.\n */\n summary: string\n /**\n * Convenience type-tag — consumers can branch on `result.kind`.\n */\n kind: 'agent-eval-rl-campaign'\n}\n\nexport async function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult> {\n const splitTag = opts.splitTag ?? 'search'\n\n // ── 1. Run the matrix ──────────────────────────────────────────────\n const campaign = await runEvalCampaign({ ...opts, splitTag })\n\n // ── 2. Extract reward signals (deterministic-first) ────────────────\n const rewardSignals = extractVerifiableRewardsFromRecords(\n campaign.runs,\n opts.verifiableReward ?? {},\n )\n\n // ── 3. Mint the scored runs once, then derive all training artifacts ──\n const scoredRuns = campaign.runs.filter((run) => runTaskScore(run) !== undefined)\n const { rows: rolloutLines } = await mintRolloutRows(scoredRuns, new InMemoryTraceStore())\n const preferences = extractPreferences(rolloutLines, {\n ...opts.preferences,\n strategy: opts.preferences?.strategy ?? 'paired-by-scenario-and-seed',\n minMargin: opts.preferences?.minMargin ?? 0.05,\n split: opts.preferences?.split ?? splitTag,\n })\n\n // ── 4. Sequential / anytime-valid interim verdict ──────────────────\n let interimConfidence: InterimReleaseConfidence | null = null\n let deltaCoverage: PairedDeltaCoverage[] = []\n const minDeltaCoverage = opts.sequential?.minDeltaCoverage ?? 1\n if (!(Number.isFinite(minDeltaCoverage) && minDeltaCoverage >= 0 && minDeltaCoverage <= 1)) {\n throw new Error(\n `runRLCampaign: sequential.minDeltaCoverage must be a finite fraction in [0, 1], got ${minDeltaCoverage}`,\n )\n }\n if (opts.report?.comparator) {\n const comparator = opts.report.comparator\n const series = collectPairedDeltaSeries(campaign.runs, campaign.failedRuns, comparator)\n deltaCoverage = series.map((s) => s.coverage)\n // Fail closed on a shrunken denominator: the recommendation can be\n // `promote_now`, so it does not get computed over the cells that happened\n // to pair. The accounting ships either way.\n const covered = series.every((s) => s.coverage.coverage >= minDeltaCoverage)\n if (covered && series.some((s) => s.deltas.length > 0)) {\n interimConfidence = evaluateInterimReleaseConfidence({\n deltaSeries: series.map(({ candidateId, deltas }) => ({ candidateId, deltas })),\n alpha: opts.sequential?.alpha,\n bound: opts.sequential?.bound,\n rope: opts.sequential?.rope ?? opts.report?.rope,\n })\n }\n }\n\n // ── 5. Standing reward-hacking hygiene ─────────────────────────────\n const rewardHacking = detectRewardHacking({\n runs: campaign.runs,\n verifiableRewardOptions: opts.verifiableReward,\n })\n\n // ── 6. Predictive validity (when outcomes are supplied) ────────────\n let predictiveValidity: RubricPredictiveValidityReport | null = null\n if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) {\n predictiveValidity = await rubricPredictiveValidity({\n runs: campaign.runs,\n outcomes: opts.outcomeStore,\n outcomeMetrics: opts.outcomeMetrics,\n })\n }\n\n // ── 7. Trainer-format export ───────────────────────────────────────\n const trainerRows: RLCampaignResult['trainerRows'] = {}\n if (opts.trainerExport?.dpo) {\n trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, {\n lines: rolloutLines,\n })\n }\n if (opts.trainerExport?.grpo) {\n trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo)\n }\n if (opts.trainerExport?.sft) {\n trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft)\n }\n\n const summary = buildSummary({\n campaign,\n preferences,\n interimConfidence,\n deltaCoverage,\n rewardHacking,\n predictiveValidity,\n })\n\n return {\n campaign,\n rewardSignals,\n preferences,\n interimConfidence,\n deltaCoverage,\n rewardHacking,\n predictiveValidity,\n trainerRows,\n summary,\n kind: 'agent-eval-rl-campaign',\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\n/**\n * Pair on (scenarioId, seed) and count what was DEALT, not only what paired.\n *\n * Every drop below used to be silent: a comparator run with no score never\n * entered the map, a candidate run with no score was skipped, and a candidate\n * cell with no comparator at the same identity was skipped. The surviving\n * deltas then flowed into `evaluateInterimReleaseConfidence`, whose\n * `recommendation.decision` can be `promote_now` — so a candidate that scored 6\n * of 26 cells could be recommended for promotion off a series of length 6, with\n * nothing in the result saying 20 cells went dark. Same defect as the promotion\n * gates, one call frame up.\n *\n * The denominator is MEASURED: a cell counts as dealt because a run for it\n * exists on either arm. Missing scores are not imputed — the caller who knows\n * the failure value of its metric writes it onto the record.\n */\nfunction collectPairedDeltaSeries(\n runs: RunRecord[],\n failedRuns: FailedRun[],\n comparator: string,\n): Array<{ candidateId: string; deltas: number[]; coverage: PairedDeltaCoverage }> {\n const cellKey = (r: { scenarioId: string; seed: number }) => `${r.scenarioId}::${r.seed}`\n // A cell that failed integrity or crashed never reaches `campaign.runs` at\n // all, so counting only the surviving records would make the very failure this\n // check exists to catch invisible. `failedRuns` carries the same\n // (variantId, scenarioId, seed) identity — it is dealt work that produced no\n // score, which is exactly what the denominator must hold.\n const dealtFromFailures = new Map<string, Set<string>>()\n for (const f of failedRuns) {\n let set = dealtFromFailures.get(f.variantId)\n if (!set) {\n set = new Set<string>()\n dealtFromFailures.set(f.variantId, set)\n }\n set.add(cellKey(f))\n }\n // Comparator side, split into what it was dealt and what it answered.\n const comparatorDealt = new Set<string>(dealtFromFailures.get(comparator) ?? [])\n const comparatorScore = new Map<string, number>()\n for (const r of runs) {\n if (r.candidateId !== comparator) continue\n const key = cellKey(r)\n comparatorDealt.add(key)\n // Ungated (`runTaskScore` is raw): this is a measurement of the paired\n // delta between candidates, not a value any trainer consumes. Gating it\n // would report a candidate as worse than it measured; the gamed run should\n // be excluded upstream instead.\n const score = runTaskScore(r)\n if (score === undefined) continue\n comparatorScore.set(key, score)\n }\n const dealtByCandidate = new Map<string, Set<string>>()\n const scoreByCandidate = new Map<string, Map<string, number>>()\n for (const [variantId, cells] of dealtFromFailures) {\n if (variantId === comparator) continue\n dealtByCandidate.set(variantId, new Set(cells))\n }\n for (const r of runs) {\n if (r.candidateId === comparator) continue\n const key = cellKey(r)\n let dealt = dealtByCandidate.get(r.candidateId)\n if (!dealt) {\n dealt = new Set<string>()\n dealtByCandidate.set(r.candidateId, dealt)\n }\n dealt.add(key)\n const score = runTaskScore(r)\n if (score === undefined) continue\n let scored = scoreByCandidate.get(r.candidateId)\n if (!scored) {\n scored = new Map<string, number>()\n scoreByCandidate.set(r.candidateId, scored)\n }\n scored.set(key, score)\n }\n\n return [...dealtByCandidate.entries()].map(([candidateId, candidateDealt]) => {\n const scored = scoreByCandidate.get(candidateId) ?? new Map<string, number>()\n const deltas: number[] = []\n let unscoredCandidate = 0\n let unscoredComparator = 0\n let unmatched = 0\n // The dealt set is the union: a cell the comparator ran and this candidate\n // never wrote a row for is still work the comparison was given.\n for (const key of new Set([...candidateDealt, ...comparatorDealt])) {\n const onCandidate = candidateDealt.has(key)\n const onComparator = comparatorDealt.has(key)\n if (!onCandidate || !onComparator) {\n unmatched += 1\n continue\n }\n const a = scored.get(key)\n const b = comparatorScore.get(key)\n if (a === undefined) unscoredCandidate += 1\n else if (b === undefined) unscoredComparator += 1\n else deltas.push(a - b)\n }\n const dealt = new Set([...candidateDealt, ...comparatorDealt]).size\n return {\n candidateId,\n deltas,\n coverage: {\n candidateId,\n dealt,\n answered: deltas.length,\n unscoredCandidate,\n unscoredComparator,\n unmatched,\n coverage: dealt === 0 ? 0 : deltas.length / dealt,\n },\n }\n })\n}\n\nfunction buildSummary(args: {\n campaign: EvalCampaignResult\n preferences: PreferenceExtractionReport\n interimConfidence: InterimReleaseConfidence | null\n deltaCoverage: PairedDeltaCoverage[]\n rewardHacking: RewardHackingReport\n predictiveValidity: RubricPredictiveValidityReport | null\n}): string {\n const c = args.campaign\n const lines = [\n `${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}…)`,\n `preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`,\n ]\n if (args.interimConfidence) {\n lines.push(\n `sequential verdict: ${args.interimConfidence.recommendation.decision}` +\n (args.interimConfidence.recommendation.candidateId\n ? ` ${args.interimConfidence.recommendation.candidateId}`\n : ''),\n )\n }\n // Never a silent 0 — a shrunken denominator has to say by how much, including\n // (especially) on the path where the verdict was refused for being shrunken.\n const shortfall = args.deltaCoverage.filter((c) => c.answered < c.dealt)\n if (shortfall.length > 0) {\n lines.push(\n `paired-delta coverage: ${shortfall\n .map((c) => `${c.candidateId} ${c.answered}/${c.dealt}`)\n .join(', ')}${args.interimConfidence ? '' : ' (sequential verdict withheld)'}`,\n )\n }\n lines.push(\n `reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`,\n )\n if (args.predictiveValidity) {\n const top = args.predictiveValidity.ranked[0]\n lines.push(\n `top-rubric: ${top?.rubric ?? 'none'} ρ=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? 'no data'})`,\n )\n }\n return lines.join(' | ')\n}\n\n// Re-export `runEvalCampaign` so consumers can pick the lower-level\n// primitive without flipping import paths.\nexport { runEvalCampaign } from '../eval-campaign'\n","/**\n * Adapters: convert measurement outputs into the canonical `RunRecord[]`\n * artifact that `replayCache`, `pairedEvalueSequence`, and\n * `rubricPredictiveValidity` consume. Two sources:\n * - `campaignToRunRecords` — the campaign substrate's per-cell results\n * (the modern path: `runCampaign` / `runImprovementLoop` → records).\n * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.\n *\n * Adapters are thin and explicit — every mandatory `RunRecord` field comes\n * from a caller-supplied context (`commitSha`, `model`, `promptHash`,\n * `configHash`) plus the cell's runtime data. The validator still rejects\n * bare-alias model strings — the caller snapshot-pins.\n */\n\nimport { campaignCellToRunRecord } from '../campaign/run-record'\nimport type { CampaignResult } from '../campaign/types'\nimport type { LayerResult, VerificationReport } from '../multi-layer-verifier'\nimport type { RunRecord, RunSplitTag } from '../run-record'\n\nexport interface AdapterContext {\n /** Logical experiment id — typically the campaign or sweep identifier. */\n experimentId: string\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** Hash of the effective prompt sent to the model. */\n promptHash: string\n /** Hash of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Default split tag. Default `'search'`. */\n splitTag?: RunSplitTag\n /** Estimated cost in USD when the source doesn't record one. */\n defaultCostUsd?: number\n}\n\n/**\n * Convert a `CampaignResult` into canonical `RunRecord[]`, one per cell.\n * Successful judged cells carry their mean judge composite and dimensions.\n * Errored or unjudged cells remain unlabeled while retaining explicit terminal\n * outcome, execution-error count, token usage, cost, and failure detail.\n * `candidateId` identifies the measured surface and defaults to the campaign\n * manifest hash.\n */\nexport function campaignToRunRecords(\n campaign: CampaignResult,\n ctx: AdapterContext & { candidateId?: string },\n): RunRecord[] {\n const splitTag = ctx.splitTag ?? 'search'\n const candidateId = ctx.candidateId ?? campaign.manifestHash\n return campaign.cells.map((cell) =>\n campaignCellToRunRecord(cell, {\n runId: cell.cellId,\n experimentId: ctx.experimentId,\n candidateId,\n model: ctx.model,\n promptHash: ctx.promptHash,\n configHash: ctx.configHash,\n commitSha: ctx.commitSha,\n splitTag,\n defaultCostUsd: ctx.defaultCostUsd,\n }),\n )\n}\n\n/**\n * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.\n * A split score is emitted only when `report.taskScore` proves the configured\n * scoring panel completed. Partial scores remain in `outcome.raw` for\n * diagnosis. Layer errors and timeouts become judge or execution telemetry;\n * only a scored `fail` layer may produce task-failure detail.\n */\nexport function verificationReportToRunRecord(\n report: VerificationReport,\n ctx: AdapterContext & { candidateId: string; scenarioId: string },\n opts: { runId?: string } = {},\n): RunRecord {\n const splitTag = ctx.splitTag ?? 'search'\n const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`\n const hasValidLayerMeasurement = report.layers.some(hasValidTaskMeasurement)\n const taskScore =\n hasValidLayerMeasurement && isValidScore(report.taskScore) ? report.taskScore : undefined\n let executionErrorCount = 0\n let judgeErrorCount = 0\n let layerErrorCount = 0\n let layerTimeoutCount = 0\n let unscoredLayerCount = 0\n\n const raw: Record<string, number> = {\n pass_count: report.passCount,\n fail_count: report.failCount,\n error_count: report.errorCount,\n skipped_count: report.skippedCount,\n duration_ms: report.durationMs,\n execution_error_count: 0,\n }\n for (const layer of report.layers) {\n if (hasValidTaskMeasurement(layer)) raw[`layer.${layer.layer}`] = layer.score\n else unscoredLayerCount++\n raw[`layer_${layer.layer}_pass`] = layer.status === 'pass' ? 1 : 0\n if (layer.status === 'error' || layer.status === 'timeout') {\n if (layer.errorSource === 'judge') judgeErrorCount++\n else executionErrorCount++\n if (layer.status === 'error') layerErrorCount++\n else layerTimeoutCount++\n }\n if (layer.diagnostics) {\n for (const [k, v] of Object.entries(layer.diagnostics)) {\n if (typeof v === 'number' && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v\n }\n }\n }\n\n raw.execution_error_count = executionErrorCount\n if (judgeErrorCount > 0) raw.judge_error_count = judgeErrorCount\n if (layerErrorCount > 0) raw.layer_error_count = layerErrorCount\n if (layerTimeoutCount > 0) raw.layer_timeout_count = layerTimeoutCount\n if (unscoredLayerCount > 0) raw.unscored_layer_count = unscoredLayerCount\n if (taskScore !== undefined) raw.blended_score = taskScore\n\n const firstScoredFailure = report.layers.find(\n (layer) => layer.status === 'fail' && hasValidTaskMeasurement(layer),\n )\n const outcome: RunRecord['outcome'] = { raw }\n if (taskScore !== undefined) {\n if (splitTag === 'holdout') outcome.holdoutScore = taskScore\n else outcome.searchScore = taskScore\n }\n\n return {\n runId,\n experimentId: ctx.experimentId,\n candidateId: ctx.candidateId,\n seed: 0,\n model: ctx.model,\n promptHash: ctx.promptHash,\n configHash: ctx.configHash,\n commitSha: ctx.commitSha,\n wallMs: report.durationMs,\n costUsd: ctx.defaultCostUsd ?? null,\n costProvenance:\n ctx.defaultCostUsd === undefined\n ? { kind: 'uncaptured', usd: null }\n : { kind: 'estimated', usd: ctx.defaultCostUsd },\n tokenUsage: { input: 0, output: 0 },\n terminalOutcome: 'succeeded',\n outcome,\n ...(firstScoredFailure\n ? {\n failureClass: 'unknown' as const,\n failureMode: `layer_${firstScoredFailure.layer}_fail`,\n }\n : {}),\n splitTag,\n scenarioId: ctx.scenarioId,\n }\n}\n\nfunction hasValidTaskMeasurement(\n layer: LayerResult,\n): layer is LayerResult & { status: 'pass' | 'fail'; score: number } {\n return (layer.status === 'pass' || layer.status === 'fail') && isValidScore(layer.score)\n}\n\nfunction isValidScore(score: unknown): score is number {\n return typeof score === 'number' && Number.isFinite(score) && score >= 0 && score <= 1\n}\n","/**\n * Simulator fidelity — score a user SIMULATOR's realism against real-user\n * trace distributions.\n *\n * Synthetic-persona evals (`PersonaConfig`-driven canonical evals, fuzz\n * user-simulator objectives) stand in for real users in most of the numbers\n * we publish. The standing threat is the Sim2Real gap: a simulator that is\n * distributionally unlike production creates \"easy mode\" and silently\n * inflates every score built on it. This module measures that gap from the\n * SAME artifact both sides already produce — `RunRecord`s — so no new\n * capture pipeline is needed:\n *\n * - `simFidelityReport` — per-feature Jensen-Shannon divergence between\n * simulated and production record distributions, collapsed into a\n * fidelity coefficient in [0,1].\n * - `easyModeCheck` — the headline academic failure mode (sim inflates\n * pass-rate over production) as its own named artifact.\n *\n * Every synthetic-persona eval result should publish its fidelity\n * coefficient alongside the score — a number from an unrepresentative\n * simulator is an unlabeled estimate. Wire-in points:\n *\n * - canonical persona evals: pass the campaign's `RunRecord`s as\n * `simulated` and intake-adapter output (`contract/intake`: OTel spans,\n * feedback tables, coding-agent sessions) as `production`\n * - the fuzz user-sim objective: use `1 - report.fidelity` as a realism\n * penalty when searching over generated personas\n * - the durable corpus (`./corpus`): both sides read straight from\n * `readCorpus` — tag sim vs production by `experimentId`\n */\n\nimport { ValidationError } from '../errors'\nimport { observedSplitScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport type { FailureClass } from '../trace/schema'\nimport type { CorpusRecord } from './corpus'\n\n/** Extracts a flat behavioral feature map from one record. `string` values\n * are categorical, `number` values are quantile-bucketed over the union of\n * both sides, `null` means the feature is absent on this record and is\n * counted explicitly as its own category (never silently dropped). */\nexport type BehaviorFeatures = (record: RunRecord) => Record<string, string | number | null>\n\n/** Reserved histogram category for `null` feature values. A capture-rate\n * difference (one side instruments a signal, the other does not) registers\n * as divergence by design: a simulator that produces no tool traces is not\n * representative of production that does. */\nexport const ABSENT_CATEGORY = '(absent)'\n\n/** Minimum non-null observations PER SIDE for a feature to enter the\n * fidelity mean. Below this the JSD estimate is sampling noise. */\nconst DEFAULT_MIN_N_PER_FEATURE = 20\n\n/** Quantile buckets used to discretize numeric features. Quartiles balance\n * resolution against per-bucket sample size at the default minN. */\nconst DEFAULT_QUANTILE_BUCKETS = 4\n\n/** Fidelity at or above this → 'representative'; below → 'skewed'.\n * 1 − 0.8 = mean JSD 0.2 ≈ distributions that mostly overlap with one\n * clearly shifted mode — the point where per-feature shifts start changing\n * which failure classes an eval can even observe. */\nconst REPRESENTATIVE_MIN_FIDELITY = 0.8\n\nconst TOP_SHIFT_COUNT = 5\n\n/**\n * Default feature set — ONLY fields verified present on both simulated and\n * production records:\n *\n * - `score`, `wall_ms`, `output_tokens` — mandatory per the `RunRecord`\n * validator (non-finite values read as absent rather than poisoning a\n * bucket).\n * - `failure_class` — optional taxonomy field; absent counted explicitly.\n * - `turn_count`, `tool_errors`, `tool_error_recovery` — derived from the\n * `outcome.raw` counters the intake adapters and eval harnesses write\n * (`turns_completed`, `assistant_messages`, `tool_errors`,\n * `turns_aborted`); absent on records whose producer did not capture\n * them, counted explicitly.\n * - `completion_length` — from the optional `CorpusRecord` trajectory\n * text; the message-length proxy when records come from the corpus.\n *\n * `RunRecord` carries event COUNTS, not event ordering, so\n * `tool_error_recovery` is a counts-only derivation: errors occurred and the\n * run still completed cleanly ('recovered') vs aborted or classified as a\n * failure ('unrecovered') — not a literal error→retry sequence check.\n */\nexport const defaultBehaviorFeatures: BehaviorFeatures = (record) => {\n const raw: Record<string, number> = record.outcome?.raw ?? {}\n const toolErrors = finiteOrNull(raw.tool_errors)\n const turnsAborted = finiteOrNull(raw.turns_aborted)\n const completion = (record as CorpusRecord).completion\n return {\n // RAW (`observedSplitScore`), deliberately: this feature vector is one\n // half of a sim-vs-production divergence measurement. Gating a gamed run to\n // 0 would move the simulated distribution toward production and report the\n // simulator as MORE faithful precisely where it is being gamed. Each split\n // is read separately rather than through `observedScore` so a non-finite\n // holdout score falls back to search instead of poisoning the bucket.\n score:\n finiteOrNull(observedSplitScore(record, 'holdout')) ??\n finiteOrNull(observedSplitScore(record, 'search')),\n failure_class: record.failureClass ?? null,\n wall_ms: finiteOrNull(record.wallMs),\n output_tokens: finiteOrNull(record.tokenUsage?.output),\n turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),\n tool_errors: toolErrors,\n tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),\n completion_length: typeof completion === 'string' ? completion.length : null,\n }\n}\n\nfunction toolErrorRecovery(\n toolErrors: number | null,\n turnsAborted: number | null,\n failureClass: FailureClass | undefined,\n): string | null {\n if (toolErrors === null) return null\n if (toolErrors === 0) return 'no-tool-errors'\n const failed =\n (turnsAborted ?? 0) > 0 || (failureClass !== undefined && failureClass !== 'success')\n return failed ? 'unrecovered' : 'recovered'\n}\n\nfunction finiteOrNull(value: unknown): number | null {\n return typeof value === 'number' && Number.isFinite(value) ? value : null\n}\n\n// ── Divergence core ──────────────────────────────────────────────────\n\n/**\n * Jensen-Shannon divergence between two categorical histograms (raw counts;\n * normalized internally). Log base 2 → bounded [0,1]: 0 = identical\n * distributions, 1 = disjoint support. Symmetric, defined even where the\n * supports differ — exactly the regime sim-vs-production comparison lives in.\n * Throws on zero-mass or negative/non-finite counts: an empty histogram has\n * no distribution and a silent 0 would read as \"perfectly representative\".\n */\nexport function jsDivergence(p: Record<string, number>, q: Record<string, number>): number {\n const keys = new Set([...Object.keys(p), ...Object.keys(q)])\n if (keys.size === 0) {\n throw new ValidationError('jsDivergence: both histograms are empty')\n }\n let pSum = 0\n let qSum = 0\n for (const key of keys) {\n const pv = p[key] ?? 0\n const qv = q[key] ?? 0\n if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) {\n throw new ValidationError(`jsDivergence: negative or non-finite count for category \"${key}\"`)\n }\n pSum += pv\n qSum += qv\n }\n if (pSum === 0 || qSum === 0) {\n throw new ValidationError('jsDivergence: a histogram with zero total mass has no distribution')\n }\n let divergence = 0\n for (const key of keys) {\n const pp = (p[key] ?? 0) / pSum\n const qp = (q[key] ?? 0) / qSum\n const m = (pp + qp) / 2\n if (pp > 0) divergence += 0.5 * pp * Math.log2(pp / m)\n if (qp > 0) divergence += 0.5 * qp * Math.log2(qp / m)\n }\n // float error can land epsilon outside [0,1]\n return Math.min(1, Math.max(0, divergence))\n}\n\n/**\n * Deterministic quantile edges over a value set (the UNION of both sides, so\n * sim and production land in the same buckets). Linear interpolation between\n * order statistics; duplicate edges from heavy ties collapse into fewer,\n * wider buckets. Returns `bucketCount - 1` edges before deduplication.\n */\nexport function quantileEdges(values: number[], bucketCount = DEFAULT_QUANTILE_BUCKETS): number[] {\n if (values.length === 0) {\n throw new ValidationError('quantileEdges: requires at least one value')\n }\n if (!Number.isInteger(bucketCount) || bucketCount < 2) {\n throw new ValidationError(\n `quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`,\n )\n }\n const sorted = [...values].sort((a, b) => a - b)\n const edges: number[] = []\n for (let k = 1; k < bucketCount; k++) {\n const pos = (k / bucketCount) * (sorted.length - 1)\n const lo = sorted[Math.floor(pos)]!\n const hi = sorted[Math.ceil(pos)]!\n edges.push(lo + (pos - Math.floor(pos)) * (hi - lo))\n }\n return [...new Set(edges)]\n}\n\n/** Stable half-open bucket label for a value against quantile edges:\n * `[-inf,e0)`, `[e0,e1)`, …, `[eLast,+inf)`. */\nexport function bucketLabel(value: number, edges: number[]): string {\n let i = 0\n while (i < edges.length && value >= edges[i]!) i++\n const lo = i === 0 ? '-inf' : String(edges[i - 1]!)\n const hi = i === edges.length ? '+inf' : String(edges[i]!)\n return `[${lo},${hi})`\n}\n\n// ── Fidelity report ──────────────────────────────────────────────────\n\nexport interface FeatureShift {\n /** Category label (a string value, a numeric bucket, or `ABSENT_CATEGORY`). */\n value: string\n /** Probability of this category among ALL simulated records (nulls included\n * via `ABSENT_CATEGORY`, so each side's shifts sum to 1). */\n pSim: number\n /** Probability among ALL production records. */\n pProd: number\n}\n\nexport interface FeatureDivergence {\n feature: string\n /** Jensen-Shannon divergence in [0,1] for this feature. */\n divergence: number\n /** Largest |pSim − pProd| categories, descending — where the sim deviates. */\n topShifts: FeatureShift[]\n /** Non-null observations on the simulated side. */\n nSim: number\n /** Non-null observations on the production side. */\n nProd: number\n}\n\nexport type FidelityVerdict = 'representative' | 'skewed' | 'insufficient-data'\n\nexport interface FidelityReport {\n perDimension: FeatureDivergence[]\n /** 1 − mean divergence over features with sufficient data. NaN when the\n * verdict is 'insufficient-data' — a 0 would read as \"maximally skewed\"\n * and silently poison downstream aggregation; check `verdict` first. */\n fidelity: number\n /** Features excluded because either side had fewer than `minNPerFeature`\n * non-null observations. Named, never silently dropped. */\n insufficientData: string[]\n /** 'representative' when fidelity >= REPRESENTATIVE_MIN_FIDELITY (0.8),\n * 'skewed' below, 'insufficient-data' when no feature met minN. */\n verdict: FidelityVerdict\n}\n\nexport interface SimFidelityOptions {\n /** Feature extractor. Defaults to `defaultBehaviorFeatures`. */\n features?: BehaviorFeatures\n /** Minimum non-null observations per side per feature. Default 20. */\n minNPerFeature?: number\n}\n\n/**\n * Compare a simulator's RunRecords against production RunRecords, feature by\n * feature. Numeric features are bucketed by deterministic quantiles of the\n * union; nulls count as an explicit `ABSENT_CATEGORY`. Throws on empty\n * inputs — \"no records\" is a wiring error, not a distribution.\n */\nexport function simFidelityReport(\n simulated: RunRecord[],\n production: RunRecord[],\n opts: SimFidelityOptions = {},\n): FidelityReport {\n if (simulated.length === 0) {\n throw new ValidationError('simFidelityReport: simulated records are empty')\n }\n if (production.length === 0) {\n throw new ValidationError('simFidelityReport: production records are empty')\n }\n const extract = opts.features ?? defaultBehaviorFeatures\n const minN = opts.minNPerFeature ?? DEFAULT_MIN_N_PER_FEATURE\n\n const simMaps = simulated.map(extract)\n const prodMaps = production.map(extract)\n\n // union of feature names in first-seen order — extractors may emit\n // different keys per record (e.g. domain-conditional features)\n const featureNames: string[] = []\n const seen = new Set<string>()\n for (const map of [...simMaps, ...prodMaps]) {\n for (const name of Object.keys(map)) {\n if (!seen.has(name)) {\n seen.add(name)\n featureNames.push(name)\n }\n }\n }\n\n const perDimension: FeatureDivergence[] = []\n const insufficientData: string[] = []\n\n for (const feature of featureNames) {\n const simVals = simMaps.map((m) => m[feature] ?? null)\n const prodVals = prodMaps.map((m) => m[feature] ?? null)\n const nSim = simVals.filter((v) => v !== null).length\n const nProd = prodVals.filter((v) => v !== null).length\n if (nSim < minN || nProd < minN) {\n insufficientData.push(feature)\n continue\n }\n const { sim, prod } = histograms(feature, simVals, prodVals)\n perDimension.push({\n feature,\n divergence: jsDivergence(sim, prod),\n topShifts: topShifts(sim, simVals.length, prod, prodVals.length),\n nSim,\n nProd,\n })\n }\n\n if (perDimension.length === 0) {\n return { perDimension, fidelity: Number.NaN, insufficientData, verdict: 'insufficient-data' }\n }\n const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length\n return {\n perDimension,\n fidelity,\n insufficientData,\n verdict: fidelity >= REPRESENTATIVE_MIN_FIDELITY ? 'representative' : 'skewed',\n }\n}\n\ntype FeatureValue = string | number | null\n\nfunction histograms(\n feature: string,\n simVals: FeatureValue[],\n prodVals: FeatureValue[],\n): { sim: Record<string, number>; prod: Record<string, number> } {\n const kinds = new Set<string>()\n for (const v of [...simVals, ...prodVals]) {\n if (v !== null) kinds.add(typeof v)\n }\n if (kinds.size > 1) {\n throw new ValidationError(\n `simFidelityReport: feature \"${feature}\" mixes string and number values — an extractor must return one kind per feature`,\n )\n }\n let toCategory: (v: string | number) => string\n if (kinds.has('number')) {\n const union: number[] = []\n for (const v of [...simVals, ...prodVals]) {\n if (v !== null) union.push(v as number)\n }\n const edges = quantileEdges(union)\n toCategory = (v) => bucketLabel(v as number, edges)\n } else {\n toCategory = (v) => v as string\n }\n const count = (vals: FeatureValue[]): Record<string, number> => {\n const hist: Record<string, number> = {}\n for (const v of vals) {\n const key = v === null ? ABSENT_CATEGORY : toCategory(v)\n hist[key] = (hist[key] ?? 0) + 1\n }\n return hist\n }\n return { sim: count(simVals), prod: count(prodVals) }\n}\n\nfunction topShifts(\n sim: Record<string, number>,\n simTotal: number,\n prod: Record<string, number>,\n prodTotal: number,\n): FeatureShift[] {\n const keys = [...new Set([...Object.keys(sim), ...Object.keys(prod)])]\n const shifts = keys.map((value) => ({\n value,\n pSim: (sim[value] ?? 0) / simTotal,\n pProd: (prod[value] ?? 0) / prodTotal,\n }))\n shifts.sort((a, b) => {\n const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd)\n return delta !== 0 ? delta : a.value.localeCompare(b.value)\n })\n return shifts.slice(0, TOP_SHIFT_COUNT)\n}\n\n// ── Easy-mode check ──────────────────────────────────────────────────\n\nexport interface EasyModeOptions {\n /** A run passes when its score (holdout, else search) >= this. Default 0.5\n * — matches the pass-threshold convention across the rl/ primitives. */\n passThreshold?: number\n /** Pass-rate gap above which the sim is flagged inflated. Default 0.1 —\n * a 10-point inflation is enough to flip most promotion gates. */\n inflationTolerance?: number\n}\n\nexport interface EasyModeReport {\n simPassRate: number\n prodPassRate: number\n /** simPassRate − prodPassRate. Positive = the simulator is easier than reality. */\n gap: number\n /** True when gap > inflationTolerance: numbers measured against this\n * simulator overstate production performance. */\n inflated: boolean\n}\n\n/**\n * The headline simulator failure mode as its own named artifact: a simulator\n * that creates \"easy mode\" inflates pass-rate relative to production, and\n * every score measured against it overstates reality. Throws on empty inputs\n * and on records carrying neither score — a silently-skipped record would\n * bias the very rate this check exists to keep honest.\n */\nexport function easyModeCheck(\n simulated: RunRecord[],\n production: RunRecord[],\n opts: EasyModeOptions = {},\n): EasyModeReport {\n if (simulated.length === 0) {\n throw new ValidationError('easyModeCheck: simulated records are empty')\n }\n if (production.length === 0) {\n throw new ValidationError('easyModeCheck: production records are empty')\n }\n const threshold = opts.passThreshold ?? 0.5\n const tolerance = opts.inflationTolerance ?? 0.1\n const passRate = (records: RunRecord[], side: string): number => {\n let passes = 0\n for (const r of records) {\n // RAW, same reason as `defaultBehaviorFeatures`: this rate exists to\n // catch a simulator that reports easier successes than production. Gating\n // would zero the inflated runs and hide the inflation being measured.\n const score =\n finiteOrNull(observedSplitScore(r, 'holdout')) ??\n finiteOrNull(observedSplitScore(r, 'search'))\n if (score === null) {\n throw new ValidationError(\n `easyModeCheck: ${side} run \"${r.runId}\" carries neither holdoutScore nor searchScore`,\n )\n }\n if (score >= threshold) passes++\n }\n return passes / records.length\n }\n const simPassRate = passRate(simulated, 'simulated')\n const prodPassRate = passRate(production, 'production')\n const gap = simPassRate - prodPassRate\n return { simPassRate, prodPassRate, gap, inflated: gap > tolerance }\n}\n","/**\n * Bradley-Terry / Elo tournament evaluation.\n *\n * For multi-candidate sweeps, comparing every candidate's score against\n * a fixed comparator wastes information — the comparator becomes a high-\n * variance reference and rank flips between near-tied middle-rank\n * candidates are dominated by noise. Pairwise tournaments fix this:\n * every (i, j) pair contributes a comparison to a Bradley-Terry MLE that\n * estimates each candidate's strength on a unified scale.\n *\n * For online updating (rolling campaigns where new candidates arrive\n * over time), we also ship classical Elo with configurable K-factor.\n *\n * References:\n * - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete\n * block designs. Biometrika, 39(3/4), 324–345.\n * - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry\n * models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm\n * used here.)\n * - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.\n *\n * This is a useful primitive because most LLM-eval communities (Chatbot\n * Arena, AlpacaEval, ELO-style ablation) have converged on pairwise\n * tournament eval as the most sample-efficient and most rank-stable\n * method when you have many candidates.\n */\n\nexport interface PairwiseOutcome {\n /** Winner candidate id. */\n winner: string\n /** Loser candidate id. */\n loser: string\n /**\n * Optional draw flag. When true, both candidates get half-credit\n * (Bradley-Terry handles draws as half-wins for each side).\n */\n draw?: boolean\n /**\n * Optional weight — useful if some pairwise comparisons are stronger\n * signals than others (e.g. a paired test with a wider score gap is\n * a more confident comparison). Default 1.\n */\n weight?: number\n}\n\nexport interface BradleyTerryRating {\n candidateId: string\n /** Latent strength θ ≥ 0 from the BT MLE. */\n strength: number\n /** Log-strength = log(θ) — interpretable on a linear scale. */\n logStrength: number\n /** Number of pairwise comparisons this candidate appears in. */\n n: number\n /** Win count (+ 0.5 per draw). */\n wins: number\n}\n\nexport interface BradleyTerryFit {\n ratings: BradleyTerryRating[]\n /** Iterations of the MM algorithm before convergence. */\n iterations: number\n /** Final maximum |θ_new - θ_old| / θ_old. */\n finalDelta: number\n converged: boolean\n}\n\n/**\n * Bradley-Terry MLE via Hunter's MM algorithm.\n *\n * Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)\n * where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.\n *\n * Returns log-strengths normalized so the smallest is 0 (any constant\n * offset is unobservable in BT — only differences are identified).\n */\nexport function fitBradleyTerry(\n outcomes: PairwiseOutcome[],\n opts: { tolerance?: number; maxIterations?: number; smoothing?: number } = {},\n): BradleyTerryFit {\n const tol = opts.tolerance ?? 1e-6\n const maxIter = opts.maxIterations ?? 256\n // Small positive default — Hunter's MM degenerates when a candidate has\n // zero wins (θ → 0 → log → -∞). 0.1 is negligible against real win counts\n // (~1 win / 10 comparisons) and keeps the iteration well-conditioned.\n // Override to 0 if the comparison set is guaranteed strongly connected.\n const smoothing = opts.smoothing ?? 0.1\n\n const candidates = new Set<string>()\n for (const o of outcomes) {\n candidates.add(o.winner)\n candidates.add(o.loser)\n }\n const ids = [...candidates].sort()\n const idx = new Map(ids.map((id, i) => [id, i]))\n const n = ids.length\n if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true }\n if (n === 1) {\n return {\n ratings: [{ candidateId: ids[0]!, strength: 1, logStrength: 0, n: 0, wins: 0 }],\n iterations: 0,\n finalDelta: 0,\n converged: true,\n }\n }\n\n // Build win matrix W[i][j] = (weighted) wins of i over j, plus half for draws.\n // Build comparison matrix N[i][j] = total weighted comparisons between i and j.\n const W: number[][] = Array.from({ length: n }, () => new Array<number>(n).fill(0))\n const N: number[][] = Array.from({ length: n }, () => new Array<number>(n).fill(0))\n for (const o of outcomes) {\n const i = idx.get(o.winner)!\n const j = idx.get(o.loser)!\n const w = o.weight ?? 1\n if (o.draw) {\n W[i]![j]! += 0.5 * w\n W[j]![i]! += 0.5 * w\n } else {\n W[i]![j]! += w\n }\n N[i]![j]! += w\n N[j]![i]! += w\n }\n\n // Per-candidate total wins.\n const winsTotal = new Array<number>(n).fill(0)\n for (let i = 0; i < n; i++) {\n for (let j = 0; j < n; j++) winsTotal[i]! += W[i]![j]!\n winsTotal[i]! += smoothing // tiny smoothing to keep θ positive\n }\n const compsTotal = new Array<number>(n).fill(0)\n for (let i = 0; i < n; i++) {\n for (let j = 0; j < n; j++) compsTotal[i]! += N[i]![j]!\n }\n\n // MM iterations.\n let theta = new Array<number>(n).fill(1)\n let iter = 0\n let delta = Infinity\n for (; iter < maxIter; iter++) {\n const newTheta = new Array<number>(n)\n for (let i = 0; i < n; i++) {\n let denom = 0\n for (let j = 0; j < n; j++) {\n if (j === i) continue\n if (N[i]![j]! === 0) continue\n denom += N[i]![j]! / (theta[i]! + theta[j]!)\n }\n newTheta[i] = denom === 0 ? theta[i]! : winsTotal[i]! / denom\n }\n // Normalize so geometric mean = 1 (numerical stability).\n let logSum = 0\n for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]!))\n const norm = Math.exp(logSum / n)\n for (let i = 0; i < n; i++) newTheta[i] = newTheta[i]! / norm\n\n delta = 0\n for (let i = 0; i < n; i++) {\n const d = Math.abs(newTheta[i]! - theta[i]!) / Math.max(1e-12, theta[i]!)\n if (d > delta) delta = d\n }\n theta = newTheta\n if (delta < tol) break\n }\n\n const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))))\n const ratings: BradleyTerryRating[] = ids.map((id, i) => ({\n candidateId: id,\n strength: theta[i]!,\n logStrength: Math.log(Math.max(1e-300, theta[i]!)) - minLog,\n n: compsTotal[i]!,\n wins: winsTotal[i]! - smoothing,\n }))\n\n return {\n ratings: ratings.sort((a, b) => b.strength - a.strength),\n iterations: iter,\n finalDelta: delta,\n converged: delta < tol,\n }\n}\n\n/**\n * Online Elo updates. Use when comparisons arrive over time and you want\n * a running rating without re-fitting the full BT MLE on every update.\n *\n * Initialize ratings to `defaultRating` (1500 by default). Each call to\n * `applyEloUpdate` mutates the map in place and returns the deltas so\n * the caller can log per-comparison rating changes.\n */\nexport interface EloOptions {\n /** Default rating for unseen candidates. Default 1500. */\n defaultRating?: number\n /** K-factor controls the step size. Default 32 (FIDE-ish). */\n kFactor?: number\n}\n\nexport function applyEloUpdate(\n ratings: Map<string, number>,\n outcome: PairwiseOutcome,\n opts: EloOptions = {},\n): { winnerDelta: number; loserDelta: number } {\n const defaultRating = opts.defaultRating ?? 1500\n const k = opts.kFactor ?? 32\n\n const rW = ratings.get(outcome.winner) ?? defaultRating\n const rL = ratings.get(outcome.loser) ?? defaultRating\n\n const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400))\n const scoreW = outcome.draw ? 0.5 : 1\n const scoreL = outcome.draw ? 0.5 : 0\n const w = outcome.weight ?? 1\n\n const winnerDelta = k * w * (scoreW - expectedW)\n const loserDelta = k * w * (scoreL - (1 - expectedW))\n\n ratings.set(outcome.winner, rW + winnerDelta)\n ratings.set(outcome.loser, rL + loserDelta)\n\n return { winnerDelta, loserDelta }\n}\n\n/**\n * Build pairwise outcomes from the campaign artifact: for every scenario\n * shared by two candidates, the higher-scoring run wins. Useful when you\n * want a tournament view of an existing campaign without an additional\n * pairwise judge call.\n */\nexport interface BuildPairwiseFromCampaignInput {\n runs: Array<{\n candidateId: string\n /** Stable identifier for the matching unit (typically scenarioId). */\n matchKey: string\n score: number\n }>\n /**\n * Tied-score margin. Below this, the comparison is a draw. Default 0\n * (no ties).\n */\n drawMargin?: number\n}\n\nexport function buildPairwiseFromCampaign(\n input: BuildPairwiseFromCampaignInput,\n): PairwiseOutcome[] {\n const drawMargin = input.drawMargin ?? 0\n const byKey = new Map<string, Array<{ candidateId: string; score: number }>>()\n for (const r of input.runs) {\n const arr = byKey.get(r.matchKey) ?? []\n arr.push({ candidateId: r.candidateId, score: r.score })\n byKey.set(r.matchKey, arr)\n }\n const outcomes: PairwiseOutcome[] = []\n for (const arr of byKey.values()) {\n for (let i = 0; i < arr.length; i++) {\n for (let j = i + 1; j < arr.length; j++) {\n const a = arr[i]!\n const b = arr[j]!\n if (a.candidateId === b.candidateId) continue\n const margin = Math.abs(a.score - b.score)\n if (margin <= drawMargin) {\n outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 })\n } else {\n const [winner, loser] = a.score > b.score ? [a, b] : [b, a]\n outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin })\n }\n }\n }\n }\n return outcomes\n}\n","/**\n * Verified-findings dataset — execution-verified gold labels as RL-ready rows.\n *\n * A replay-verify batch re-executes a labeled trajectory prefix inside the\n * original docker image and checks, at the gold \"incorrect\" step k, whether\n * the recorded failure reproduces (arm A) and whether a generated fix makes\n * it vanish (arm B). That turns an annotation into an *executed* label: the\n * verdict is a returncode/signature comparison, not a rater's opinion.\n *\n * This module joins three artifact families into one row per replayed case:\n *\n * 1. the batch report (`batch-report.json` — per-case verdicts, fix arms),\n * 2. the gold label corpus (`*-labels.json` — incorrect step annotations),\n * 3. the normalized trajectory (`normalized/<trajId>/steps.json` — the\n * action/observation sequence the agent actually took).\n *\n * The emitted `VerifiedFindingRow` carries the trajectory prefix up to k,\n * the gold label, the execution verdict with its evidence (exit codes,\n * failure signature, prefix divergences), the fix arm when present, and\n * per-row provenance (label/steps/report sha256s, docker images, run ids).\n * Rows are trainer input for step-level localizer/critic models; the reward\n * is deterministic because execution decided it.\n *\n * Join discipline: every missing or inconsistent join throws — a dataset\n * built from partially joined artifacts would silently train on wrong\n * labels. The batch report is authoritative for fix outcomes (per-case\n * `replay-verdict.json` files are written before the fix arm completes);\n * per-case files contribute prefix-divergence detail and run ids only, and\n * are cross-checked against the report where they overlap.\n */\n\nimport { createHash } from 'node:crypto'\nimport { readFileSync } from 'node:fs'\nimport { join } from 'node:path'\n\nexport const VERIFIED_FINDING_SCHEMA = 'agent-eval/verified-finding@0'\n\n// ── Input shapes (parsed artifacts) ─────────────────────────────────\n\n/** One case row from a replay-verify `batch-report.json`. */\nexport interface ReplayBatchCase {\n corpus: string\n trajId: string\n image: string\n cwd: string\n cwdSource: string\n k: number\n stepCount: number\n goldIncorrectSteps: number[]\n recordedReturncodeAtK: number\n derivedImage: string | null\n signature: string | null\n status: string\n error: string | null\n prefixExecuted: number\n prefixDivergences: number\n prefixDivergencePct: number\n prefixReturncodeMismatches: number\n prefixUnknownExpectations: number\n armAExit: number | null\n armAReturncodeMatch: boolean\n armASignatureMatch: boolean\n /** Batch verdict: prefix divergence within tolerance AND arm A reproduced the recorded returncode at k. */\n replayed: boolean\n fix: ReplayBatchFix | null\n wallMs: number\n}\n\nexport interface ReplayBatchFix {\n attempted: boolean\n sampledOut: boolean\n command: string | null\n llmError: string | null\n armBExit: number | null\n failureVanished: boolean | null\n}\n\nexport interface ReplayBatchReport {\n generatedAt: string\n cases: ReplayBatchCase[]\n}\n\n/** Gold label entry for one trajectory (CodeTraceBench annotation format). */\nexport interface GoldLabelEntry {\n traj_id: string\n solved: boolean\n step_count: number\n agent?: string\n model?: string\n task_name?: string\n difficulty?: string\n incorrect_stages: Array<{ stage_id: number; incorrect_step_ids: number[] }>\n}\n\n/** One step from a normalized trajectory `steps.json`. */\nexport interface NormalizedStep {\n step_id: number\n action: string\n observation?: string | null\n}\n\nexport interface PrefixDivergence {\n step: number\n /** `unknown-expectation` marks a step the recording carries no returncode\n * for: it could not be confirmed, so it is not agreement. */\n kind: 'returncode-mismatch' | 'unknown-expectation'\n /** null exactly when `kind` is `unknown-expectation`. */\n expectedReturncode: number | null\n actualExit: number\n}\n\n/** Optional extract from a per-case `replay-verdict.json` (arm A detail only). */\nexport interface CaseVerdictDetail {\n k: number\n prefixExecuted: number\n recordedReturncode: number\n signatureBasis: string | null\n prefixDivergences: PrefixDivergence[]\n armACommand: string | null\n runIds: { original: string | null; armA: string | null }\n}\n\n// ── Row schema ──────────────────────────────────────────────────────\n\nexport interface TrajectoryStep {\n stepId: number\n action: string\n observation: string | null\n /** True when the observation was cut at `maxObservationChars`; `observationChars` keeps the original length. */\n observationTruncated: boolean\n observationChars: number\n}\n\nexport type FixOutcome = 'flipped' | 'not-flipped' | 'generation-failed' | 'not-attempted'\n\nexport interface VerifiedFindingFix {\n outcome: FixOutcome\n command: string | null\n llmError: string | null\n armBExit: number | null\n failureVanished: boolean | null\n}\n\nexport interface VerifiedFindingRow {\n schema: typeof VERIFIED_FINDING_SCHEMA\n /** `<runId>/<corpus>/<trajId>` — unique across batches. */\n caseId: string\n corpus: string\n trajId: string\n task: {\n agent: string | null\n model: string | null\n taskName: string | null\n difficulty: string | null\n solved: boolean\n stepCount: number\n }\n gold: {\n /** The verified gold step — the earliest replayable incorrect step. */\n stepK: number\n /** The exact command the agent ran at step k (never truncated — it is the labeled object). */\n actionAtK: string\n /** Incorrect steps the batch considered replay targets (submit-step golds excluded). */\n goldIncorrectSteps: number[]\n /** Every incorrect step in the label entry, across stages. */\n labelIncorrectSteps: number[]\n recordedReturncodeAtK: number\n }\n /** Prefix context 1..k — post-k steps are excluded so a trainer never sees the future. */\n trajectory: {\n window: { start: number; end: number }\n steps: TrajectoryStep[]\n }\n verification: {\n reproduced: boolean\n /** Arm A output also contained the recorded error substring (or returncode-only basis matched). */\n signatureStrict: boolean\n signatureBasis: string | null\n signature: string | null\n prefixExecuted: number\n prefixDivergences: number\n prefixDivergencePct: number\n prefixReturncodeMismatches: number\n /** Steps the recording carries no returncode for. Nonzero here means part\n * of the prefix replay was never confirmed against the recording. */\n prefixUnknownExpectations: number\n prefixDivergenceDetail: PrefixDivergence[] | null\n armAExit: number | null\n armAReturncodeMatch: boolean\n armACommand: string | null\n wallMs: number\n }\n fix: VerifiedFindingFix\n provenance: {\n runId: string\n batchGeneratedAt: string\n batchReportSha256: string\n labelsPath: string\n labelsSha256: string\n stepsPath: string\n stepsSha256: string\n image: string\n derivedImage: string | null\n cwd: string\n cwdSource: string\n originalRunId: string | null\n armARunId: string | null\n }\n}\n\nexport interface VerifiedFindingsSummary {\n rows: number\n reproduced: number\n /** Reproduced AND arm A matched the failure signature — the batch report's headline strict rate.\n * Row-level `verification.signatureStrict` is raw arm A evidence and can be true on a\n * non-reproduced case (signature matched but the prefix diverged past tolerance). */\n signatureStrict: number\n fix: Record<FixOutcome, number>\n byCorpus: Record<string, { rows: number; reproduced: number; fixFlipped: number }>\n}\n\n// ── Pure join ───────────────────────────────────────────────────────\n\nconst DEFAULT_MAX_OBSERVATION_CHARS = 4000\n\nexport interface BuildVerifiedFindingRowArgs {\n batchCase: ReplayBatchCase\n label: GoldLabelEntry\n steps: NormalizedStep[]\n runId: string\n batchGeneratedAt: string\n batchReportSha256: string\n labelsPath: string\n labelsSha256: string\n stepsPath: string\n stepsSha256: string\n detail?: CaseVerdictDetail\n maxObservationChars?: number\n}\n\nfunction fail(caseId: string, message: string): never {\n throw new Error(`verified-findings: ${caseId}: ${message}`)\n}\n\nfunction deriveFixOutcome(caseId: string, batchCase: ReplayBatchCase): VerifiedFindingFix {\n const fix = batchCase.fix\n if (fix === null) {\n if (batchCase.replayed) {\n fail(\n caseId,\n 'replayed case has no fix record — the batch always records the fix arm for replayed cases',\n )\n }\n return {\n outcome: 'not-attempted',\n command: null,\n llmError: null,\n armBExit: null,\n failureVanished: null,\n }\n }\n const base = {\n command: fix.command,\n llmError: fix.llmError,\n armBExit: fix.armBExit,\n failureVanished: fix.failureVanished,\n }\n if (fix.command !== null) {\n if (fix.failureVanished === null) {\n fail(\n caseId,\n 'fix command present but failureVanished missing — arm B verdict was never recorded',\n )\n }\n return { outcome: fix.failureVanished ? 'flipped' : 'not-flipped', ...base }\n }\n if (fix.llmError !== null) return { outcome: 'generation-failed', ...base }\n if (!fix.attempted || fix.sampledOut) return { outcome: 'not-attempted', ...base }\n fail(caseId, 'unrecognized fix record state (attempted, no command, no llmError)')\n}\n\nfunction truncateObservation(\n observation: string | null | undefined,\n maxChars: number,\n): Pick<TrajectoryStep, 'observation' | 'observationTruncated' | 'observationChars'> {\n if (observation === null || observation === undefined) {\n return { observation: null, observationTruncated: false, observationChars: 0 }\n }\n if (observation.length <= maxChars) {\n return { observation, observationTruncated: false, observationChars: observation.length }\n }\n return {\n observation: observation.slice(0, maxChars),\n observationTruncated: true,\n observationChars: observation.length,\n }\n}\n\n/**\n * Join one batch case with its gold label and trajectory into a row.\n * Throws on any join inconsistency — never emits a partially joined row.\n */\nexport function buildVerifiedFindingRow(args: BuildVerifiedFindingRowArgs): VerifiedFindingRow {\n const { batchCase, label, steps, detail } = args\n const caseId = `${args.runId}/${batchCase.corpus}/${batchCase.trajId}`\n const maxObservationChars = args.maxObservationChars ?? DEFAULT_MAX_OBSERVATION_CHARS\n\n if (batchCase.status !== 'ok') {\n fail(\n caseId,\n `case status is '${batchCase.status}' (error: ${batchCase.error ?? 'none'}) — only ok cases join`,\n )\n }\n if (label.traj_id !== batchCase.trajId) {\n fail(caseId, `label traj_id '${label.traj_id}' does not match the case`)\n }\n if (label.step_count !== batchCase.stepCount) {\n fail(caseId, `label step_count ${label.step_count} != case stepCount ${batchCase.stepCount}`)\n }\n if (steps.length !== batchCase.stepCount) {\n fail(caseId, `steps.json has ${steps.length} steps, case expects ${batchCase.stepCount}`)\n }\n for (let i = 0; i < steps.length; i++) {\n const step = steps[i]!\n if (step.step_id !== i + 1) {\n fail(caseId, `steps.json is not contiguous 1..n: index ${i} has step_id ${step.step_id}`)\n }\n }\n const k = batchCase.k\n if (k < 1 || k > batchCase.stepCount) {\n fail(caseId, `gold step k=${k} is outside 1..${batchCase.stepCount}`)\n }\n if (!batchCase.goldIncorrectSteps.includes(k)) {\n fail(\n caseId,\n `gold step k=${k} is not in goldIncorrectSteps [${batchCase.goldIncorrectSteps.join(', ')}]`,\n )\n }\n const labelIncorrectSteps = [\n ...new Set(label.incorrect_stages.flatMap((s) => s.incorrect_step_ids)),\n ].sort((a, b) => a - b)\n for (const goldStep of batchCase.goldIncorrectSteps) {\n if (!labelIncorrectSteps.includes(goldStep)) {\n fail(\n caseId,\n `case gold step ${goldStep} is absent from the label's incorrect steps — label/report mismatch`,\n )\n }\n }\n if (detail !== undefined) {\n if (detail.k !== k) fail(caseId, `per-case verdict k=${detail.k} != report k=${k}`)\n if (detail.prefixExecuted !== batchCase.prefixExecuted) {\n fail(\n caseId,\n `per-case verdict prefixExecuted=${detail.prefixExecuted} != report ${batchCase.prefixExecuted}`,\n )\n }\n if (detail.recordedReturncode !== batchCase.recordedReturncodeAtK) {\n fail(\n caseId,\n `per-case verdict recordedReturncode=${detail.recordedReturncode} != report ${batchCase.recordedReturncodeAtK}`,\n )\n }\n if (detail.prefixDivergences.length !== batchCase.prefixDivergences) {\n fail(\n caseId,\n `per-case verdict lists ${detail.prefixDivergences.length} prefix divergences != report ${batchCase.prefixDivergences}`,\n )\n }\n const unknown = detail.prefixDivergences.filter((d) => d.kind === 'unknown-expectation').length\n if (unknown !== batchCase.prefixUnknownExpectations) {\n fail(\n caseId,\n `per-case verdict lists ${unknown} unknown-expectation steps != report ${batchCase.prefixUnknownExpectations}`,\n )\n }\n }\n\n const stepAtK = steps[k - 1]!\n const trajectorySteps: TrajectoryStep[] = steps.slice(0, k).map((step) => ({\n stepId: step.step_id,\n action: step.action,\n ...truncateObservation(step.observation, maxObservationChars),\n }))\n\n return {\n schema: VERIFIED_FINDING_SCHEMA,\n caseId,\n corpus: batchCase.corpus,\n trajId: batchCase.trajId,\n task: {\n agent: label.agent ?? null,\n model: label.model ?? null,\n taskName: label.task_name ?? null,\n difficulty: label.difficulty ?? null,\n solved: label.solved,\n stepCount: batchCase.stepCount,\n },\n gold: {\n stepK: k,\n actionAtK: stepAtK.action,\n goldIncorrectSteps: [...batchCase.goldIncorrectSteps].sort((a, b) => a - b),\n labelIncorrectSteps,\n recordedReturncodeAtK: batchCase.recordedReturncodeAtK,\n },\n trajectory: {\n window: { start: 1, end: k },\n steps: trajectorySteps,\n },\n verification: {\n reproduced: batchCase.replayed,\n signatureStrict: batchCase.armASignatureMatch,\n signatureBasis: detail?.signatureBasis ?? null,\n signature: batchCase.signature,\n prefixExecuted: batchCase.prefixExecuted,\n prefixDivergences: batchCase.prefixDivergences,\n prefixDivergencePct: batchCase.prefixDivergencePct,\n prefixReturncodeMismatches: batchCase.prefixReturncodeMismatches,\n prefixUnknownExpectations: batchCase.prefixUnknownExpectations,\n prefixDivergenceDetail: detail?.prefixDivergences ?? null,\n armAExit: batchCase.armAExit,\n armAReturncodeMatch: batchCase.armAReturncodeMatch,\n armACommand: detail?.armACommand ?? null,\n wallMs: batchCase.wallMs,\n },\n fix: deriveFixOutcome(caseId, batchCase),\n provenance: {\n runId: args.runId,\n batchGeneratedAt: args.batchGeneratedAt,\n batchReportSha256: args.batchReportSha256,\n labelsPath: args.labelsPath,\n labelsSha256: args.labelsSha256,\n stepsPath: args.stepsPath,\n stepsSha256: args.stepsSha256,\n image: batchCase.image,\n derivedImage: batchCase.derivedImage,\n cwd: batchCase.cwd,\n cwdSource: batchCase.cwdSource,\n originalRunId: detail?.runIds.original ?? null,\n armARunId: detail?.runIds.armA ?? null,\n },\n }\n}\n\nexport function summarizeVerifiedFindings(rows: VerifiedFindingRow[]): VerifiedFindingsSummary {\n const summary: VerifiedFindingsSummary = {\n rows: rows.length,\n reproduced: 0,\n signatureStrict: 0,\n fix: { flipped: 0, 'not-flipped': 0, 'generation-failed': 0, 'not-attempted': 0 },\n byCorpus: {},\n }\n for (const row of rows) {\n if (row.verification.reproduced) summary.reproduced++\n if (row.verification.reproduced && row.verification.signatureStrict) summary.signatureStrict++\n summary.fix[row.fix.outcome]++\n let corpus = summary.byCorpus[row.corpus]\n if (corpus === undefined) {\n corpus = { rows: 0, reproduced: 0, fixFlipped: 0 }\n summary.byCorpus[row.corpus] = corpus\n }\n corpus.rows++\n if (row.verification.reproduced) corpus.reproduced++\n if (row.fix.outcome === 'flipped') corpus.fixFlipped++\n }\n return summary\n}\n\nexport function verifiedFindingsToJsonl(rows: VerifiedFindingRow[]): string {\n return rows.map((row) => JSON.stringify(row)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── Filesystem loader ───────────────────────────────────────────────\n\nexport interface VerifiedFindingsCorpusSource {\n labelsPath: string\n /** Directory containing `normalized/<trajId>/steps.json`. */\n preparedDir: string\n}\n\nexport interface VerifiedFindingsSource {\n batchReportPath: string\n /** Batch run identifier embedded in every caseId, e.g. 'run2-20260802'. */\n runId: string\n /** Corpus name (as it appears in the batch report) → label + trajectory locations. */\n corpora: Record<string, VerifiedFindingsCorpusSource>\n /** Batch run directory holding `<corpus>--<trajId>/replay-verdict.json`; when set, every case must have one. */\n runDir?: string\n maxObservationChars?: number\n}\n\nexport interface VerifiedFindingsDataset {\n rows: VerifiedFindingRow[]\n summary: VerifiedFindingsSummary\n provenance: {\n runId: string\n batchReportPath: string\n batchReportSha256: string\n batchGeneratedAt: string\n corpora: Record<string, { labelsPath: string; labelsSha256: string; preparedDir: string }>\n }\n}\n\nfunction sha256(buffer: Buffer): string {\n return createHash('sha256').update(buffer).digest('hex')\n}\n\nfunction readJson(path: string, what: string): { value: unknown; sha256: string } {\n let buffer: Buffer\n try {\n buffer = readFileSync(path)\n } catch (error) {\n throw new Error(\n `verified-findings: cannot read ${what} at ${path}: ${(error as Error).message}`,\n )\n }\n try {\n return { value: JSON.parse(buffer.toString('utf8')), sha256: sha256(buffer) }\n } catch (error) {\n throw new Error(\n `verified-findings: ${what} at ${path} is not valid JSON: ${(error as Error).message}`,\n )\n }\n}\n\ninterface CaseVerdictFile {\n k: number\n prefixExecuted: number\n recordedReturncode: number\n signatureBasis?: string | null\n prefixDivergences?: PrefixDivergence[]\n armA?: { command?: string | null } | null\n runIds?: { original?: string | null; armA?: string | null } | null\n}\n\nconst PREFIX_DIVERGENCE_KINDS = new Set(['returncode-mismatch', 'unknown-expectation'])\n\nfunction loadCaseVerdictDetail(runDir: string, batchCase: ReplayBatchCase): CaseVerdictDetail {\n const path = join(runDir, `${batchCase.corpus}--${batchCase.trajId}`, 'replay-verdict.json')\n const parsed = readJson(path, `per-case verdict for ${batchCase.trajId}`).value as CaseVerdictFile\n const divergences = parsed.prefixDivergences ?? []\n // A record without a kind came from a comparison that could not tell a\n // confirmed match from an unverifiable step, so the whole file is unusable.\n for (const divergence of divergences) {\n if (!PREFIX_DIVERGENCE_KINDS.has(divergence.kind)) {\n throw new Error(\n `verified-findings: ${path} step ${divergence.step} has divergence kind ` +\n `'${divergence.kind}'; expected one of ${[...PREFIX_DIVERGENCE_KINDS].join(', ')}`,\n )\n }\n }\n return {\n k: parsed.k,\n prefixExecuted: parsed.prefixExecuted,\n recordedReturncode: parsed.recordedReturncode,\n signatureBasis: parsed.signatureBasis ?? null,\n prefixDivergences: divergences,\n armACommand: parsed.armA?.command ?? null,\n runIds: {\n original: parsed.runIds?.original ?? null,\n armA: parsed.runIds?.armA ?? null,\n },\n }\n}\n\n/**\n * Load a replay-verify batch and join it into verified-finding rows.\n * Every case in the report must join: an unresolvable corpus, a missing\n * label entry, or a missing trajectory throws instead of dropping the row.\n */\nexport function loadVerifiedFindingsDataset(\n source: VerifiedFindingsSource,\n): VerifiedFindingsDataset {\n const report = readJson(source.batchReportPath, 'batch report')\n const parsedReport = report.value as ReplayBatchReport\n if (!Array.isArray(parsedReport.cases) || parsedReport.cases.length === 0) {\n throw new Error(`verified-findings: batch report at ${source.batchReportPath} has no cases`)\n }\n if (typeof parsedReport.generatedAt !== 'string' || parsedReport.generatedAt.length === 0) {\n throw new Error(\n `verified-findings: batch report at ${source.batchReportPath} has no generatedAt`,\n )\n }\n\n const labelCache = new Map<string, { sha256: string; byTrajId: Map<string, GoldLabelEntry> }>()\n const corporaProvenance: VerifiedFindingsDataset['provenance']['corpora'] = {}\n\n const resolveCorpus = (corpus: string) => {\n const config = source.corpora[corpus]\n if (config === undefined) {\n throw new Error(\n `verified-findings: batch report references corpus '${corpus}' but no labels/preparedDir was configured for it`,\n )\n }\n let cached = labelCache.get(corpus)\n if (cached === undefined) {\n const labels = readJson(config.labelsPath, `labels for corpus '${corpus}'`)\n const entries = labels.value as GoldLabelEntry[]\n if (!Array.isArray(entries)) {\n throw new Error(\n `verified-findings: labels for corpus '${corpus}' at ${config.labelsPath} are not an array`,\n )\n }\n const byTrajId = new Map<string, GoldLabelEntry>()\n for (const entry of entries) {\n if (byTrajId.has(entry.traj_id)) {\n throw new Error(\n `verified-findings: labels for corpus '${corpus}' contain duplicate traj_id '${entry.traj_id}'`,\n )\n }\n byTrajId.set(entry.traj_id, entry)\n }\n cached = { sha256: labels.sha256, byTrajId }\n labelCache.set(corpus, cached)\n corporaProvenance[corpus] = {\n labelsPath: config.labelsPath,\n labelsSha256: labels.sha256,\n preparedDir: config.preparedDir,\n }\n }\n return { config, ...cached }\n }\n\n const rows: VerifiedFindingRow[] = []\n for (const batchCase of parsedReport.cases) {\n const { config, sha256: labelsSha256, byTrajId } = resolveCorpus(batchCase.corpus)\n const label = byTrajId.get(batchCase.trajId)\n if (label === undefined) {\n throw new Error(\n `verified-findings: ${source.runId}/${batchCase.corpus}/${batchCase.trajId}: no label entry in ${config.labelsPath}`,\n )\n }\n const stepsPath = join(config.preparedDir, 'normalized', batchCase.trajId, 'steps.json')\n const stepsFile = readJson(stepsPath, `trajectory steps for ${batchCase.trajId}`)\n const steps = stepsFile.value as NormalizedStep[]\n if (!Array.isArray(steps)) {\n throw new Error(`verified-findings: trajectory steps at ${stepsPath} are not an array`)\n }\n const detail =\n source.runDir === undefined ? undefined : loadCaseVerdictDetail(source.runDir, batchCase)\n rows.push(\n buildVerifiedFindingRow({\n batchCase,\n label,\n steps,\n runId: source.runId,\n batchGeneratedAt: parsedReport.generatedAt,\n batchReportSha256: report.sha256,\n labelsPath: config.labelsPath,\n labelsSha256,\n stepsPath,\n stepsSha256: stepsFile.sha256,\n detail,\n maxObservationChars: source.maxObservationChars,\n }),\n )\n }\n rows.sort((a, b) => a.caseId.localeCompare(b.caseId))\n\n return {\n rows,\n summary: summarizeVerifiedFindings(rows),\n provenance: {\n runId: source.runId,\n batchReportPath: source.batchReportPath,\n batchReportSha256: report.sha256,\n batchGeneratedAt: parsedReport.generatedAt,\n corpora: corporaProvenance,\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2EA,eAAsB,mBACpB,MAC0B;CAC1B,MAAM,KAAK,KAAK,MAAM;EAAC;EAAG;EAAG;EAAG;EAAG;EAAG;CAAE;CACxC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,WAAW,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAE7C,MAAM,SAA4B,CAAC;CACnC,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,cAA8C,CAAC;EACrD,MAAM,YAAsB,CAAC;EAC7B,IAAI,cAAc;EAClB,IAAI,gBAAgB;EACpB,KAAK,MAAM,YAAY,KAAK,WAAW;GACrC,MAAM,MAAM,SAAS,cAAc,YAAY,KAAK,UAAU,QAAQ,QAAQ;GAC9E,MAAM,SAAmB,CAAC;GAC1B,IAAI,SAAS;GACb,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,KAAK;IAC7B,MAAM,QAAQ,MAAM,KAAK,OAAO,IAAI;KAAE;KAAU;KAAG,KAAK;IAAE,CAAC;IAC3D,OAAO,KAAK,KAAK;IACjB,IAAI,SAAS,eAAe;IAC5B,UAAU,KAAK,KAAK;IACpB,IAAI,SAAS,eAAe;IAC5B;GACF;GACA,MAAM,QAAQ,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;GACzD,YAAY,KAAK;IAAE,YAAY;IAAK,WAAW;IAAO;IAAQ,OAAO,OAAO;GAAO,CAAC;EACtF;EACA,MAAM,YAAY,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,UAAU,MAAM;EACrF,MAAM,WACJ,UAAU,SAAS,IACf,IACA,UAAU,QAAQ,GAAG,MAAM,KAAK,IAAI,cAAc,GAAG,CAAC,KAAK,UAAU,SAAS;EACpF,OAAO,KAAK;GACV;GACA;GACA,UAAU,cAAc,KAAK,IAAI,GAAG,aAAa;GACjD,KAAK,KAAK,KAAK,QAAQ;GACvB,GAAG,UAAU;GACb;EACF,CAAC;CACH;CAEA,MAAM,aAAa,OAAO,MAAM,MAAM,EAAE,YAAY,aAAa,CAAC,EAAE,KAAK;CACzE,MAAM,OAAO,SAAS,SAAS,SAAS,MAAM;CAE9C,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;EACtC,MAAM,KAAK,OAAO,IAAI,EAAE,CAAE;EAC1B,MAAM,KAAK,OAAO,EAAE,CAAE;EACtB,MAAM,KAAK,OAAO,IAAI,EAAE,CAAE;EAC1B,MAAM,KAAK,OAAO,EAAE,CAAE;EACtB,SAAU,KAAK,MAAM,KAAM,KAAK;CAClC;CAGA,OAAO;EAAE;EAAQ;EAAY,gBAFN,SAAS,IAAI,IAAI,OAAO;CAEH;AAC9C;;;;;;AAwBA,SAAgB,wBACd,GACA,GACA,OAA4E,CAAC,GACxD;CACrB,MAAM,OAAO,KAAK,cAAc;CAChC,MAAM,YAAY,KAAK,sBAAsB;CAC7C,MAAM,MAAM,QACV,KAAK,MACL,EAAE,OAAO,SAAS,UAAU,MAAM,YAAY,KAAK,SAAS,KAAK,SAAS,CAAC,GAC3E,EAAE,OAAO,SAAS,UAAU,MAAM,YAAY,KAAK,SAAS,KAAK,SAAS,CAAC,CAC7E;CAEA,MAAM,OAAoC,CAAC;CAC3C,KAAK,MAAM,MAAM,EAAE,QAAQ;EACzB,MAAM,KAAK,EAAE,OAAO,MAAM,MAAM,EAAE,MAAM,GAAG,CAAC;EAC5C,IAAI,CAAC,IAAI;EACT,MAAM,SAAS,GAAG,YAAY,KAAK,MAAM,EAAE,SAAS;EACpD,MAAM,SAAS,GAAG,YAAY,KAAK,MAAM,EAAE,SAAS;EACpD,MAAM,MAAM,gBAAgB,QAAQ,WAAW,MAAM,GAAG;EACxD,MAAM,MAAM,gBAAgB,QAAQ,WAAW,MAAM,GAAG;EACxD,KAAK,KAAK;GACR,GAAG,GAAG;GACN,WAAW,GAAG,YAAY,GAAG;GAC7B,MAAM,IAAI;GACV,OAAO,IAAI;GACX,MAAM,IAAI;GACV,OAAO,IAAI;EACb,CAAC;CACH;CAEA,MAAM,YAAY,EAAE,iBAAiB,EAAE;CACvC,MAAM,kBACJ,EAAE,eAAe,QAAQ,EAAE,eAAe,OACtC,EAAE,aAAa,EAAE,aACjB;CAIN,MAAM,YAAY,KAAK,QAAQ,GAAG,MAAM,IAAI,EAAE,WAAW,CAAC,IAAI,KAAK,IAAI,GAAG,KAAK,MAAM;CACrF,IAAI;CACJ,IAAI,KAAK,IAAI,SAAS,IAAI,OAAQ,KAAK,IAAI,SAAS,IAAI,KAAM,UAAU;MACnE,IAAI,YAAY,KAAK,YAAY,GAAG,UAAU;MAC9C,IAAI,YAAY,KAAK,YAAY,GAAG,UAAU;MAC9C,UAAU;CAEf,MAAM,YACJ,oBAAoB,UAAU,QAAQ,CAAC,EAAE,eAAe,UAAU,QAAQ,CAAC,OAC1E,oBAAoB,OAAO,wBAAwB,oBAAoB;CAE1E,OAAO;EAAE;EAAM;EAAW;EAAiB;EAAS;CAAU;AAChE;;AAGA,SAAgB,WAAW,OAAwB,YAAY,IAAoB;CACjF,OAAO,MAAM,OAAO,MAAM,MAAM,EAAE,YAAY,SAAS,CAAC,EAAE,KAAK;AACjE;AAIA,SAAS,gBACP,IACA,WACA,YACA,KAC+B;CAC/B,IAAI,GAAG,SAAS,GAAG,OAAO;EAAE,KAAK,GAAG,MAAM;EAAG,MAAM,GAAG,MAAM;CAAE;CAC9D,MAAM,UAAU,IAAI,MAAc,SAAS;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAAK;EAClC,IAAI,MAAM;EACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,OAAO,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;EAC1E,QAAQ,KAAK,MAAM,GAAG;CACxB;CACA,QAAQ,MAAM,GAAG,MAAM,IAAI,CAAC;CAC5B,MAAM,QAAQ,IAAI;CAClB,OAAO;EACL,KAAK,QAAQ,KAAK,MAAO,QAAQ,IAAK,SAAS;EAC/C,MAAM,QAAQ,KAAK,IAAI,YAAY,GAAG,KAAK,MAAM,IAAI,QAAQ,KAAK,SAAS,IAAI,CAAC;CAClF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC9JA,eAAsB,gBAAgB,MAAqD;CACzF,MAAM,SAA8B,CAAC;CACrC,KAAK,MAAM,UAAU,KAAK,SAAS;EACjC,MAAM,IAAI,MAAM,KAAK,YAAY,MAAM;EACvC,OAAO,KAAK;GACV,UAAU,OAAO;GACjB,MAAM,OAAO;GACb,OAAO,EAAE;GACT,SAAS,EAAE;GACX,KAAK,EAAE;GACP,SAAS,EAAE;EACb,CAAC;CACH;CACA,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;CACzD,MAAM,WAAW,OAAO,UAAU,IAAI,YAAY,MAAM,IAAI;CAC5D,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAO,EAAE,QAAQ,EAAE,QAAQ,IAAI,CAAE;CAChE,OAAO;EAAE,aAAa,KAAK;EAAa,QAAQ;EAAQ;EAAU;CAAK;AACzE;;AAqBA,eAAsB,QAAW,MAAkE;CACjG,IAAI,KAAK,KAAK,GAAG,MAAM,IAAI,gBAAgB,wBAAwB;CACnE,MAAM,WAAgB,CAAC;CACvB,MAAM,SAAmB,CAAC;CAC1B,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,GAAG,KAAK;EAC/B,MAAM,IAAI,MAAM,KAAK,OAAO,CAAC;EAC7B,SAAS,KAAK,CAAC;EACf,OAAO,KAAK,MAAM,KAAK,QAAQ,CAAC,CAAC;CACnC;CACA,IAAI,YAAY;CAChB,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,IAAI,OAAO,KAAM,OAAO,YAAa,YAAY;CACzF,MAAM,YAAY,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CAC7D,OAAO;EACL,MAAM,SAAS;EACf,WAAW,OAAO;EAClB;EACA;EACA;CACF;AACF;;;;;AA0BA,eAAsB,gBACpB,MACmC;CACnC,IAAI,KAAK,KAAK,GAAG,MAAM,IAAI,gBAAgB,gCAAgC;CAC3E,MAAM,WAAgB,CAAC;CACvB,MAAM,YAAoC,CAAC;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,GAAG,KAAK;EAC/B,MAAM,IAAI,MAAM,KAAK,OAAO,CAAC;EAC7B,SAAS,KAAK,CAAC;EACf,MAAM,MAAM,KAAK,UAAU,CAAC;EAC5B,UAAU,QAAQ,UAAU,QAAQ,KAAK;CAC3C;CACA,IAAI,SAAS;CACb,IAAI,MAAM;CACV,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,SAAS,GAC3C,IAAI,IAAI,KAAK;EACX,MAAM;EACN,SAAS;CACX;CAEF,MAAM,iBAAiB,SAAS,MAAM,MAAM,KAAK,UAAU,CAAC,MAAM,MAAM,KAAK,SAAS;CACtF,OAAO;EACL;EACA,WAAW,MAAM,KAAK;EACtB;EACA;EACA;CACF;AACF;AAeA,SAAgB,eAAe,QAAgD;CAC7E,MAAM,aAAiC,CAAC;CACxC,KAAK,MAAM,KAAK,QAKd,IAAI,CAJc,OAAO,MACtB,MACC,MAAM,KAAK,EAAE,QAAQ,EAAE,QAAQ,EAAE,SAAS,EAAE,UAAU,EAAE,OAAO,EAAE,QAAQ,EAAE,QAAQ,EAAE,MAE5E,GAAG,WAAW,KAAK,CAAC;CAEnC,OAAO,WAAW,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;AAClD;AAIA,SAAS,YAAY,QAAqC;CAIxD,MAAM,KAAK,OAAO,KAAK,MAAM,KAAK,IAAI,KAAK,IAAI,OAAO,EAAE,IAAI,CAAC,CAAC;CAC9D,MAAM,KAAK,OAAO,KAAK,MAAM,EAAE,KAAK;CACpC,MAAM,IAAI,GAAG;CACb,MAAM,KAAK,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3C,MAAM,KAAK,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3C,IAAI,MAAM;CACV,IAAI,MAAM;CACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,QAAQ,GAAG,KAAM,OAAO,GAAG,KAAM;EACjC,QAAQ,GAAG,KAAM,OAAO;CAC1B;CACA,OAAO,QAAQ,IAAI,IAAI,MAAM;AAC/B;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACrIA,eAAsB,sBACpB,OACA,OAAkC,CAAC,GACA;CACnC,MAAM,MAAM,KAAK,OAAO;CACxB,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,QAAQ,KAAK,cAAc;CAEjC,IAAI,CAAC,MAAM,aAAa,CAAC,MAAM,cAC7B,MAAM,IAAI,gBACR,0EACF;CAEF,MAAM,YACJ,MAAM,aAAc,MAAM,QAAQ,IAAI,MAAM,UAAU,KAAK,MAAM,MAAM,aAAc,MAAM,CAAC,CAAC,CAAC;CAChG,IAAI,UAAU,WAAW,MAAM,UAAU,QACvC,MAAM,IAAI,gBACR,2CAA2C,UAAU,OAAO,eAAe,MAAM,UAAU,QAC7F;CAIF,MAAM,aAAa,MAAM,QAAQ,IAAI,MAAM,UAAU,KAAK,MAAM,MAAM,QAAQ,CAAC,CAAC,CAAC;CACjF,MAAM,aAAa,MAAM,QAAQ,IAAI,UAAU,KAAK,MAAM,MAAM,QAAQ,CAAC,CAAC,CAAC;CAE3E,MAAM,cAAc,MAAM,UAAU,KAAK,GAAG,OAAO;EACjD,YAAY,MAAM,WAAW,CAAC;EAC9B,eAAe,WAAW;EAC1B,gBAAgB,WAAW;EAC3B,OAAO,WAAW,KAAM,WAAW;EACnC,QAAQ;CACV,EAAE;CAGF,MAAM,QAAQ,YAAY,QAAQ,MAAM,EAAE,iBAAiB,SAAS,EAAE,kBAAkB,KAAK;CAC7F,IAAI,MAAM,SAAS,GACjB,OAAO;EACL;EACA,YAAY;GAAE,GAAG;GAAG,GAAG;EAAE;EACzB,aAAa;EACb,WAAW;EACX,wBAAwB;EACxB,QAAQ,mCAAmC,MAAM,OAAO;EACxD,GAAG,MAAM;CACX;CAKF,MAAM,aAAa,mBAFD,MAAM,KAAK,MAAM,EAAE,aAES,GAD5B,MAAM,KAAK,MAAM,EAAE,cACoB,CAAC;CAC1D,MAAM,SAAS,MAAM,KAAK,MAAM,EAAE,KAAK;CACvC,MAAM,eAAe,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CACrD,MAAM,SAAS,aAAa,KAAK,MAAM,aAAa,SAAS,CAAC;CAC9D,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CAOxD,MAAM,EAAE,YAAY,kBADJ,MAAM,KAAK,MAAM,KAAK,IAAI,GAAG,KAAK,IAAI,MAAM,IAAI,KAAK,IAAI,EAAE,KAAK,IAAI,CAAC,CAAC,CAC1C,GAAG,GAAG;CAClD,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;EACrC,MAAM,IAAI,MAAM;EAChB,MAAM,MAAM,YAAY,WAAW,MAAM,EAAE,eAAe,EAAE,UAAU;EACtE,IAAI,OAAO,GAAG,YAAY,IAAI,CAAE,SAAS,QAAQ;CACnD;CAEA,MAAM,yBAAyB,WAAW,IAAI,OAAO,UAAU,CAAC;CAOhE,OAAO;EACL;EACA;EACA,aAAa;EACb,WAAW;EACX;EACA,QAZa,yBACX,YAAY,WAAW,EAAE,QAAQ,CAAC,EAAE,KAAK,IAAI,mBAAmB,OAAO,QAAQ,CAAC,EAAE,KAAK,kBACvF,WAAW,KAAK,MACd,uCAAuC,WAAW,EAAE,QAAQ,CAAC,EAAE,KAC/D,8CAA8C,OAAO,QAAQ,CAAC,EAAE;EASpE,GAAG,MAAM;CACX;AACF;;;;;;;AAUA,SAAgB,gBACd,aACA,UAAiD,GAAG,MAAM,GAAG,EAAE,IAAK,IAAI,KAAM,GAAA,CAAI,SAAS,EAAE,KACpE;CACzB,OAAO;EACL,MAAM;EACN,MAAM,UAAU;GACd,IAAI,SAAS,SAAS;GACtB,YAAY,SAAS,IAAI,MAAM;IAC7B,MAAM,cAAc,OAAO,IAAI,CAAC;IAChC,MAAM,KAAK,IAAI,OAAO,MAAM,YAAY,EAAE,EAAE,MAAM,GAAG;IACrD,SAAS,OAAO,QAAQ,IAAI,WAAW;GACzC,CAAC;GACD,OAAO;IAAE,GAAG;IAAU;GAAO;EAC/B;CACF;AACF;;;;;;AAOA,SAAgB,aACd,gBACA,MACyB;CACzB,IAAI,IAAI,SAAS;CACjB,MAAM,YAAoB;EACxB,IAAK,IAAI,eAAgB;EACzB,IAAI,IAAI;EACR,IAAI,KAAK,KAAK,IAAK,MAAM,IAAK,IAAI,CAAC;EACnC,KAAK,IAAI,KAAK,KAAK,IAAK,MAAM,GAAI,IAAI,EAAE;EACxC,SAAS,IAAK,MAAM,QAAS,KAAK;CACpC;CACA,OAAO;EACL,MAAM;EACN,MAAM,UAAU;GACd,MAAM,YAAY,eAAe,SAAS,QAAQ,GAAG;GACrD,OAAO;IAAE,GAAG;IAAU,QAAQ;GAAU;EAC1C;CACF;AACF;;;;;;AAOA,SAAgB,uBACd,QACA,WAAgC,UACP;CACzB,OAAO;EACL,MAAM;EACN,MAAM,UAAU;GACd,MAAM,SACJ,aAAa,WAAW,GAAG,OAAO,GAAG,SAAS,WAAW,GAAG,SAAS,OAAO,GAAG;GACjF,OAAO;IAAE,GAAG;IAAU;GAAO;EAC/B;CACF;AACF;AAIA,SAAS,YAAY,GAAmB;CACtC,OAAO,EAAE,QAAQ,uBAAuB,MAAM;AAChD;;;;;;;;;;;;;;;ACtPA,SAAgB,oBAAoB,MAAwC;CAI1E,iBAAiB,MAAM,kBAAkB;CACzC,MAAM,EAAE,WAAW,KAAK;CACxB,IAAI,WAAW,QAAQ,CAAC,OAAO,SAAS,MAAM,GAAG,OAAO;CACxD,OAAO;AACT;;AAGA,SAAgB,oBAAoB,MAAkC;CACpE,OAAO,KAAK,QAAQ,mBAAmB;AACzC;AAgEA,SAAS,KAAK,OAAyC,KAAa,MAA+B;CACjG,MAAM,WAAW,MAAM,IAAI,GAAG;CAC9B,IAAI,aAAa,KAAA,GAAW,MAAM,IAAI,KAAK,CAAC,IAAI,CAAC;MAC5C,SAAS,KAAK,IAAI;AACzB;AAEA,SAAS,gBAAgB,OAAsD;CAC7E,MAAM,4BAAY,IAAI,IAAiC;CACvD,MAAM,wBAAQ,IAAI,IAAiC;CACnD,KAAK,MAAM,QAAQ,OAAO;EACxB,KAAK,WAAW,KAAK,YAAY,IAAI;EACrC,KAAK,OAAO,KAAK,QAAQ,IAAI;CAC/B;CACA,OAAO;EAAE;EAAW;CAAM;AAC5B;;;;;;;;;;;;;;;;;;;AAoBA,SAAS,kBAAkB,OAAwB,IAAwB;CACzE,MAAM,WAAW,MAAM,UAAU,IAAI,EAAE,KAAK,CAAC;CAC7C,MAAM,OAAO,MAAM,MAAM,IAAI,EAAE,KAAK,CAAC;CACrC,IAAI,SAAS,SAAS,GAAG,OAAO;EAAE,MAAM;EAAa,OAAO,SAAS;CAAO;CAC5E,MAAM,QAAQ,SAAS;CACvB,IAAI,UAAU,KAAA,GAAW;EACvB,IAAI,KAAK,MAAM,SAAS,SAAS,KAAK,GACpC,OAAO;GAAE,MAAM;GAAa,OAAO,IAAI,KAAK,QAAQ,SAAS,SAAS,KAAK,CAAC,CAAC;EAAO;EAEtF,OAAO;GAAE,MAAM;GAAY,MAAM;EAAM;CACzC;CACA,IAAI,KAAK,SAAS,GAAG,OAAO;EAAE,MAAM;EAAa,OAAO,KAAK;CAAO;CACpE,MAAM,OAAO,KAAK;CAClB,OAAO,SAAS,KAAA,IAAY,EAAE,MAAM,UAAU,IAAI;EAAE,MAAM;EAAY,MAAM;CAAK;AACnF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoDA,SAAgB,yBACd,OACA,OACA,SACA,aACA,SACmB;CACnB,IAAI,YAAY,KAAA,KAAa,YAAY,MACvC,MAAM,IAAI,MACR,GAAG,YAAY,SAAS,MAAM,YAAY,YAAY,iBAAiB,YAAY,QAAQ,wDAC7F;CAEF,MAAM,QAAQ,gBAAgB,QAAQ,KAAK;CAC3C,MAAM,QAA2B;EAC/B,UAAU,CAAC;EACX,YAAY;EACZ,gBAAgB;EAChB,WAAW,CAAC;CACd;CACA,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAA6B,CAAC;EACpC,IAAI,YAAY;EAChB,KAAK,MAAM,MAAM,MAAM,IAAI,GAAG;GAC5B,MAAM,aAAa,kBAAkB,OAAO,EAAE;GAC9C,IAAI,WAAW,SAAS,WACtB,MAAM,IAAI,MACR,GAAG,YAAY,SAAS,qCAAqC,GAAG,qDAClE;GAEF,IAAI,WAAW,SAAS,aAAa;IACnC,YAAY;IACZ,IAAI,CAAC,MAAM,UAAU,MAAM,UAAU,MAAM,OAAO,EAAE,GAClD,MAAM,UAAU,KAAK;KAAE;KAAI,aAAa,WAAW;IAAM,CAAC;IAE5D;GACF;GACA,MAAM,KAAK,WAAW,IAAI;EAC5B;EACA,KAAK,MAAM,QAAQ,OAAO;GACxB,iBAAiB,MAAM,YAAY,QAAQ;GAC3C,UAAU,IAAI;EAChB;EACA,IAAI,WAAW;GACb,MAAM;GACN;EACF;EACA,IAAI,MAAM,KAAK,mBAAmB,GAAG;GACnC,MAAM;GACN;EACF;EACA,MAAM,SAAS,KAAK,IAAI;CAC1B;CACA,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,yBACd,OACA,OACA,SACA,aACA,SACK;CACL,MAAM,QAAQ,yBAAyB,OAAO,OAAO,SAAS,aAAa,OAAO;CAClF,IAAI,MAAM,iBAAiB,GAAG;EAC5B,MAAM,QAAQ,MAAM,UAAU,KAAK,MAAM,GAAG,EAAE,GAAG,IAAI,EAAE,YAAY,cAAc,CAAC,CAAC,KAAK,IAAI;EAC5F,QAAQ,KACN,IAAI,YAAY,SAAS,YAAY,MAAM,eAAe,YAAY,MAAM,6OAI9E;CACF;CACA,OAAO,MAAM;AACf;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC3MA,MAAM,0BAAkD;CACtD,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;;;;;;;;;AAiBA,eAAsB,UACpB,SACA,SACA,SACyB;CACzB,MAAM,WAAW,yBACf,UACC,MAAM,CAAC,EAAE,aAAa,EAAE,aAAa,GACtC,SACA,uBACF;CACA,MAAM,MAAsB,CAAC;CAC7B,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,CAAC,cAAc,gBAAgB,QAAQ,YAAY,MAAM,QAAQ,IAAI;GACzE,QAAQ,QAAQ,QAAQ,SAAS,EAAE,WAAW,CAAC;GAC/C,QAAQ,QAAQ,QAAQ,SAAS,EAAE,aAAa,CAAC;GACjD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,WAAW,CAAC;GACnD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,aAAa,CAAC;EACvD,CAAC;EACD,IAAI,iBAAiB,gBACnB,MAAM,IAAI,MACR,0BAA0B,EAAE,YAAY,KAAK,EAAE,cAAc,gCAC/D;EAEF,IAAI,KAAK;GACP,QAAQ;GACR;GACA;GACA,QAAQ,EAAE;GACV,MAAM;IACJ,YAAY,EAAE;IACd,iBAAiB,EAAE;IACnB,mBAAmB,EAAE;IACrB,aAAa,EAAE;IACf,eAAe,EAAE;IACjB,aAAa,EAAE,KAAK;IACpB,eAAe,EAAE,KAAK;GACxB;EACF,CAAC;CACH;CACA,OAAO;AACT;;AAGA,SAAgB,WAAW,MAA8B;CACvD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;;;;;;;;;;;;;;;;;;AAkDA,eAAsB,WACpB,OACA,SAC0B;CAC1B,OAAO,kBAAkB,OAAO,OAAO;AACzC;AAEA,eAAe,kBACb,OACA,SAC0B;CAC1B,MAAM,0BAAU,IAAI,IAAiC;CACrD,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,gBAAgB,MAAM,OAAO,GAAG;EACrC,MAAM,MAAM,QAAQ,IAAI,KAAK,KAAK,WAAW,KAAK,CAAC;EACnD,IAAI,KAAK,IAAI;EACb,QAAQ,IAAI,KAAK,KAAK,aAAa,GAAG;CACxC;CAEA,MAAM,OAAwB,CAAC;CAC/B,KAAK,MAAM,CAAC,YAAY,UAAU,QAAQ,QAAQ,GAAG;EACnD,IAAI,MAAM,WAAW,GAAG;EACxB,MAAM,SAA6D,CAAC;EACpE,KAAK,MAAM,QAAQ,OAAO;GACxB,MAAM,SAAS,oBAAoB,IAAI;GACvC,IAAI,WAAW,MAAM;GACrB,OAAO,KAAK;IAAE;IAAM;GAAO,CAAC;EAC9B;EAGA,IAAI,OAAO,SAAS,GAAG;EACvB,MAAM,UAAU,MAAM,QAAQ,IAC5B,OAAO,KAAK,EAAE,WAAW,QAAQ,QAAQ,QAAQ,SAAS,KAAK,MAAM,CAAC,CAAC,CACzE;EACA,MAAM,SAAS,QAAQ;EACvB,IAAI,QAAQ,MAAM,UAAU,UAAU,MAAM,GAC1C,MAAM,IAAI,MACR,yBAAyB,WAAW,qDACtC;EAEF,MAAM,cAAc,MAAM,QAAQ,IAChC,OAAO,KAAK,EAAE,WAAW,QAAQ,QAAQ,QAAQ,aAAa,KAAK,MAAM,CAAC,CAAC,CAC7E;EACA,MAAM,UAAU,OAAO,KAAK,EAAE,aAAa,MAAM;EACjD,MAAM,SAAS,OAAO,KAAK,EAAE,WAAW,KAAK,MAAM;EACnD,KAAK,KAAK;GACR;GACA;GACA;GACA;GACA,MAAM;IACJ;IACA,GAAG,YAAY;IACf,YAAY,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,QAAQ;GAC3D;EACF,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAgB,YAAY,MAA+B;CACzD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;;;;;;;;;;;;;;;;;;;AAwCA,eAAsB,UACpB,OACA,SACyB;CACzB,OAAO,iBAAiB,OAAO,OAAO;AACxC;AAEA,eAAe,iBACb,OACA,SACyB;CACzB,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,0BAA0B,QAAQ,2BAA2B;CACnE,IAAI,CAAC,OAAO,SAAS,uBAAuB,GAC1C,MAAM,IAAI,MAAM,wCAAwC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,QAAQ,OAAO;EAIxB,iBAAiB,MAAM,YAAY;EACnC,IAAI,oBAAoB,IAAI,GAAG;EAC/B,IAAI,CAAC,gBAAgB,MAAM,OAAO,GAAG;EACrC,MAAM,QAAQ,oBAAoB,IAAI;EACtC,IAAI,UAAU,QAAQ,SAAS,yBAAyB;EACxD,IAAI,CAAC,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,UAAU,MACpF;EAEF,IAAI,CAAC,QAAQ,IAAI,GAAG;EACpB,MAAM,SAAS,QAAQ,WAAW,IAAI;EACtC,MAAM,CAAC,QAAQ,cAAc,MAAM,QAAQ,IAAI,CAC7C,QAAQ,QAAQ,QAAQ,SAAS,KAAK,MAAM,CAAC,GAC7C,QAAQ,QAAQ,QAAQ,aAAa,KAAK,MAAM,CAAC,CACnD,CAAC;EACD,MAAM,WAAqC,CAAC;EAC5C,IAAI,QAAQ,SAAS,KAAK;GAAE,MAAM;GAAU,SAAS;EAAO,CAAC;EAC7D,SAAS,KAAK;GAAE,MAAM;GAAQ,SAAS;EAAO,CAAC;EAC/C,SAAS,KAAK;GAAE,MAAM;GAAa,SAAS;EAAW,CAAC;EACxD,KAAK,KAAK;GACR;GACA,MAAM;IACJ,OAAO,KAAK;IACZ,aAAa,KAAK,gBAAgB;IAClC,YAAY,KAAK,KAAK;IACtB;IACA,OAAO,KAAK,OAAO;GACrB;EACF,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAgB,WAAW,MAA8B;CACvD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;;;;;;;;;;;;;;;;;;;;AA0DA,eAAsB,UACpB,SACA,SACA,SACyB;CACzB,MAAM,WAAW,gBAAgB,SAAS,OAAO;CACjD,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,SAAS,MAAM,QAAQ,QAAQ,QAAQ,SAAS,EAAE,WAAW,CAAC;EACpE,MAAM,gBAAgB,QAAQ,WAC1B,MAAM,QAAQ,QAAQ,QAAQ,SAAS,EAAE,aAAa,EAAE,eAAe,CAAC,IACxE,CAAC;EACL,MAAM,iBAA2B,CAAC;EAClC,KAAK,MAAM,UAAU,eACnB,eAAe,KAAK,MAAM,QAAQ,QAAQ,QAAQ,WAAW,EAAE,aAAa,MAAM,CAAC,CAAC;EAEtF,MAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,WAAW,EAAE,aAAa,EAAE,YAAY,CAAC;EAC1F,MAAM,eAAe,MAAM,QAAQ,QACjC,QAAQ,WAAW,EAAE,eAAe,EAAE,cAAc,CACtD;EACA,KAAK,KAAK;GACR;GACA;GACA;GACA;GACA;GACA,cAAc,EAAE;GAChB,gBAAgB,EAAE;GAClB,aAAa,EAAE;GACf,MAAM;IACJ,aAAa,EAAE;IACf,eAAe,EAAE;IACjB,iBAAiB,EAAE;GACrB;EACF,CAAC;CACH;CACA,OAAO;AACT;;;;;;;;;AAUA,SAAS,uBAAuB,MAAyB,oBAAmC;CAC1F,MAAM,KAAK,KAAK;CAChB,IAAI,KAAK,WAAW,QAAQ,KAAA,GAC1B,MAAM,IAAI,MACR,uBAAuB,GAAG,kBAAkB,KAAK,WAAW,IAAI,qFAClE;CAEF,IAAI,KAAK,UAAU,KAAA,KAAa,KAAK,MAAM,WAAW,GACpD,MAAM,IAAI,MACR,uBAAuB,GAAG,8EAC5B;CAEF,IAAI,KAAK,QAAQ,cACf,MAAM,IAAI,MACR,uBAAuB,GAAG,sFAC5B;CAEF,IAAI,uBAAuB,KAAA,KAAa,KAAK,MAAM,UAAU,oBAC3D,MAAM,IAAI,MACR,uBAAuB,GAAG,OAAO,KAAK,MAAM,OAAO,4BAA4B,mBAAmB,wGACpG;AAEJ;AAEA,MAAM,0BAAkD;CACtD,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;AASA,SAAS,gBACP,SACA,SACqB;CACrB,OAAO,yBACL,UACC,MAAM,CAAC,EAAE,aAAa,EAAE,aAAa,GACtC,SACA,0BACC,SAAS,uBAAuB,MAAM,QAAQ,kBAAkB,CACnE;AACF;AAEA,SAAgB,WAAW,MAA8B;CACvD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;AAaA,MAAM,kCAA0D;CAC9D,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;;;;AAYA,SAAgB,mBAAmB,aAA2B,SAAqC;CAOjG,MAAM,OANW,yBACf,cACC,MAAM,CAAC,EAAE,KAAK,GACf,SACA,+BAEwC,CAAC,CAAC,KAAK,OAAO;EACtD,OAAO,EAAE;EACT,QAAQ,EAAE;EACV,WAAW,EAAE;EACb,QAAQ,EAAE;EACV,aAAa,EAAE;EACf,QAAQ,EAAE,UAAU;CACtB,EAAE;CACF,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;AAEA,SAAS,gBACP,MACA,SACS;CACT,IAAI,QAAQ,gBAAgB,KAAA,GAAW,OAAO,QAAQ,YAAY,SAAS,KAAK,KAAK,KAAK;CAC1F,OAAO,gBAAgB,MAAM,OAAO;AACtC;;;ACpgBA,MAAM,qCAAqB,IAAI,IAAa;CAHnB;CAAQ;CAAO;AAGkB,CAAC;AAE3D,SAAgB,uBAAuB,OAAiC;CACtE,IAAI,CAAC,MAAM,QAAQ,KAAK,KAAK,MAAM,WAAW,GAC5C,MAAM,IAAI,MAAM,sEAAsE;CAGxF,MAAM,UAA2B,CAAC;CAClC,MAAM,uBAAO,IAAI,IAAmB;CACpC,KAAK,MAAM,UAAU,OAAO;EAC1B,IAAI,CAAC,mBAAmB,IAAI,MAAM,GAChC,MAAM,IAAI,MACR,sCAAsC,KAAK,UAAU,MAAM,EAAE,0CAC/D;EAEF,MAAM,gBAAgB;EACtB,IAAI,KAAK,IAAI,aAAa,GACxB,MAAM,IAAI,MACR,oCAAoC,KAAK,UAAU,aAAa,EAAE,oCACpE;EAEF,KAAK,IAAI,aAAa;EACtB,QAAQ,KAAK,aAAa;CAC5B;CACA,OAAO;AACT;AAyEA,SAAS,SAAS,IAAgD;CAChE,OAAO,CAAC,GAAG,IAAI,IAAI,GAAG,QAAQ,MAAmB,OAAO,MAAM,YAAY,EAAE,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK;AACjG;AAEA,SAAS,mBAAmB,QAA+B;CACzD,IAAI,OAAO,WAAW,GACpB,OAAO;EAAE,GAAG;EAAG,MAAM;EAAM,QAAQ;EAAM,KAAK;EAAM,KAAK;EAAM,KAAK;CAAK;CAE3E,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,IAAI,OAAO;CACjB,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACjD,MAAM,MAAM,KAAK,MAAM,IAAI,CAAC;CAC5B,MAAM,SAAS,IAAI,MAAM,KAAK,OAAO,MAAM,KAAM,OAAO,QAAS,IAAI,OAAO;CAC5E,MAAM,WAAW,OAAO,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,IAAI;CACnE,OAAO;EAAE;EAAG;EAAM;EAAQ,KAAK,OAAO;EAAK,KAAK,OAAO,IAAI;EAAK,KAAK,KAAK,KAAK,QAAQ;CAAE;AAC3F;AAEA,SAAS,sBAAsB,OAA4C;CACzE,MAAM,SAAuC;EAAE,QAAQ;EAAG,KAAK;EAAG,SAAS;EAAG,QAAQ;CAAE;CACxF,IAAI,QAAQ;CACZ,IAAI,SAAS;CACb,IAAI,OAAO;CACX,IAAI,sBAAsB;CAC1B,MAAM,UAAoB,CAAC;CAC3B,KAAK,MAAM,QAAQ,OAAO;EACxB,OAAO,KAAK,KAAK,UAAU;EAC3B,SAAS,KAAK,KAAK,aAAa;EAChC,UAAU,KAAK,KAAK,cAAc;EAClC,IAAI,KAAK,KAAK,QAAQ,MAAM;OACvB,QAAQ,KAAK,KAAK;EACvB,MAAM,KAAK,oBAAoB,IAAI;EACnC,IAAI,OAAO,MAAM,QAAQ,KAAK,EAAE;CAClC;CACA,OAAO;EACL,SAAS,MAAM;EACf,eAAe,QAAQ;EACvB;EACA,QAAQ,mBAAmB,OAAO;EAClC,QAAQ,SAAS,MAAM,KAAK,MAAM,EAAE,OAAO,KAAK,CAAC;EACjD,cAAc,SAAS,MAAM,KAAK,MAAM,EAAE,OAAO,WAAW,CAAC;EAC7D,YAAY,SAAS,MAAM,KAAK,MAAM,EAAE,OAAO,cAAc,CAAC;EAC9D,aAAa;GAAE,OAAO;GAAO,QAAQ;EAAO;EAC5C,cAAc;EACd;CACF;AACF;;;;;;;;AASA,eAAsB,eACpB,OACA,SACA,QACA,aAC0B;CAC1B,IAAI,MAAM,WAAW,GACnB,MAAM,IAAI,MAAM,yEAAyE;CAE3F,MAAM,UAAU,uBAAuB,OAAO,YAAY,KAAA,IAAY,CAAC,KAAK,IAAI,OAAO,OAAO;CAC9F,MAAM,QAAgC,CAAC;CACvC,MAAM,YAAoD,CAAC;CAE3D,IAAI,QAAQ,SAAS,MAAM,GAAG;EAC5B,MAAM,OAAO,MAAM,WAAW,OAAO,OAAO;EAC5C,YAAY,QAAQ,KAAK,MAAM;EAC/B,MAAM,sBAAsB,YAAY,IAAI;EAC5C,UAAU,OAAO,KAAK;CACxB;CACA,IAAI,QAAQ,SAAS,KAAK,GAAG;EAC3B,MAAM,OAAO,MAAM,UAAU,OAAO,OAAO;EAC3C,YAAY,OAAO,KAAK,MAAM;EAC9B,MAAM,qBAAqB,WAAW,IAAI;EAC1C,UAAU,MAAM,KAAK;CACvB;CACA,IAAI,QAAQ,SAAS,KAAK,GAAG;EAC3B,IAAI,CAAC,aACH,MAAM,IAAI,MAAM,yEAAyE;EAE3F,MAAM,OAAO,MAAM,UAAU,YAAY,SAAS,YAAY,SAAS,EAAE,MAAM,CAAC;EAChF,YAAY,OAAO,KAAK,MAAM;EAC9B,MAAM,qBAAqB,WAAW,IAAI;EAC1C,UAAU,MAAM,KAAK;CACvB;CACA,IAAI,CAAC,OAAO,KAAK,KAAK,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,QAAQ,KAAK,KAAK,SAAS,QAAQ,CAAC,GACzF,MAAM,IAAI,MAAM,6CAA6C;CAG/D,MAAM,WAA8B;EAClC,GAAG;EACH;EACA;EACA,OAAO,sBAAsB,KAAK;CACpC;CACA,MAAM,mBAAmB,GAAG,KAAK,UAAU,UAAU,MAAM,CAAC,EAAE;CAC9D,MAAM,kBAAkB,oBAAoB,QAAQ;CACpD,OAAO;EAAE;EAAU;CAAM;AAC3B;AAEA,SAAS,YAAY,QAAuB,MAAoB;CAC9D,IAAI,SAAS,GACX,MAAM,IAAI,MAAM,8BAA8B,OAAO,oCAAoC;AAE7F;AAEA,SAAS,IAAI,GAAmB;CAC9B,OAAO,IAAI,IAAI,IAAA,CAAK,QAAQ,CAAC,EAAE;AACjC;AAEA,SAAS,KAAK,OAA8B;CAC1C,OAAO,UAAU,OAAO,QAAQ,MAAM,QAAQ,CAAC;AACjD;;AAGA,SAAgB,oBAAoB,GAA8B;CAChE,MAAM,IAAI,EAAE;CACZ,MAAM,QAAQ,EAAE,WAAW;CAC3B,MAAM,aAAc;EAAC;EAAU;EAAO;EAAW;CAAQ,CAAC,CACvD,KAAK,MAAM,SAAS,EAAE,MAAM,EAAE,OAAO,GAAG,IAAI,IAAI,EAAE,OAAO,KAAK,KAAK,EAAE,EAAE,CAAC,CACxE,KAAK,IAAI;CACZ,MAAM,WACJ,EAAE,sBAAsB,IACpB,aAAa,EAAE,oBAAoB,sCACnC;CACN,MAAM,gBAAgB,EAAE,OAAO,SAAS;CACxC,OAAO;EACL,cAAc,EAAE,KAAK,MAAM,EAAE,QAAQ;EACrC;EACA,eAAe,EAAE,OAAO,kBAAkB,EAAE,aAAa,kBAAkB,EAAE;EAC7E;EACA;EACA,eAAe,EAAE,OAAO,OAAO,gBAAgB,kCAAkC;EACjF,iBAAiB,EAAE,OAAO;EAC1B,sBAAsB,EAAE,OAAO;EAC/B;EACA;EACA,iCAAiC,EAAE;EACnC,yBAAyB,EAAE;EAC3B,kBAAkB,EAAE,QAAQ,KAAK,MAAM,GAAG,EAAE,IAAI,EAAE,UAAU,MAAM,EAAE,OAAO,CAAC,CAAC,KAAK,IAAI;EACtF;EACA;EACA;EACA;EACA,OAAO,EAAE,OAAO,EAAE,UAAU,KAAK,EAAE,OAAO,IAAI,EAAE,YAAY,KAAK,EAAE,OAAO,MAAM,EAAE,SAAS,KAAK,EAAE,OAAO,GAAG,EAAE,SAAS,KAAK,EAAE,OAAO,GAAG,EAAE,SAAS,KAAK,EAAE,OAAO,GAAG;EACpK;EACA;EACA,iBAAiB,EAAE,OAAO,KAAK,IAAI;EACnC,yCAAyC,EAAE,aAAa,OAAO;EAC/D,kBAAkB,EAAE,WAAW,KAAK,IAAI;EACxC,iBAAiB,EAAE,YAAY,MAAM,QAAQ,EAAE,YAAY,OAAO,oBAAoB,EAAE,aAAa,QAAQ,CAAC,IAAI;EAClH;EACA;EACA,0BAA0B,EAAE,cAAc,sBAAsB;EAChE,YAAY,EAAE,cAAc,QAAQ,QAAQ,KAAK,+BAA+B,EAAE,cAAc,yBAAyB,QAAQ;EACjI;EACA;EACA,EAAE;EACF;EACA;EACA,EAAE;EACF;EACA;EACA,EAAE;EACF;EACA;EACA;EACA;CACF,CAAC,CAAC,KAAK,IAAI;AACb;;;;;;;;;;;;;;;;;;;;;;;;;AC/QA,SAAgB,eAAe,SAAyB,YAAwC;CAC9F,UAAU,QAAQ,UAAU,GAAG,EAAE,WAAW,KAAK,CAAC;CAClD,MAAM,WAAW,WAAW,UAAU,IAAI,WAAW,UAAU,IAAI,CAAC;CACpE,MAAM,OAAO,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACjD,MAAM,QAAkB,CAAC;CACzB,IAAI,WAAW;CACf,IAAI,UAAU;CACd,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,KAAK,IAAI,EAAE,KAAK,GAAG;GACrB;GACA;EACF;EACA,KAAK,IAAI,EAAE,KAAK;EAChB,MAAM,KAAK,KAAK,UAAU,CAAC,CAAC;EAC5B;CACF;CACA,IAAI,MAAM,SAAS,GAAG,eAAe,YAAY,GAAG,MAAM,KAAK,IAAI,EAAE,GAAG;CACxE,OAAO;EAAE;EAAU;EAAS,OAAO,SAAS,SAAS;CAAS;AAChE;;AAGA,SAAgB,WAAW,YAAoC;CAC7D,IAAI,CAAC,WAAW,UAAU,GAAG,OAAO,CAAC;CACrC,MAAM,MAAsB,CAAC;CAC7B,KAAK,MAAM,QAAQ,aAAa,YAAY,MAAM,CAAC,CAAC,MAAM,IAAI,GAC5D,IAAI,KAAK,KAAK,GAAG,IAAI,KAAK,KAAK,MAAM,IAAI,CAAiB;CAE5D,OAAO;AACT;;;;;;AAOA,SAAS,SAAS,GAAgC;CAChD,MAAM,IAAI,cAAc,CAAC;CACzB,OAAO,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,IAAI,IAAI;AAC3D;;;;;;;;;;;;;;AAwBA,eAAsB,uBACpB,YACA,QACA,OAAuB,CAAC,GACE;CAC1B,IAAI,UAAU,WAAW,UAAU,CAAC,CAAC,QAClC,MAAM,OAAO,EAAE,WAAW,YAAY,OAAO,EAAE,eAAe,QACjE;CACA,IAAI,KAAK,QAAQ,UAAU,QAAQ,QAAQ,MAAM,KAAK,OAAQ,SAAS,EAAE,QAAQ,CAAC;CAClF,UAAU,QAAQ,QAAQ,MAAM,SAAS,CAAC,MAAM,IAAI;CACpD,IAAI,KAAK,YAAY,MACnB,UAAU,QAAQ,QAAQ,MAAM;EAC9B,MAAM,SAAS,SAAS,CAAC;EACzB,OAAO,WAAW,QAAQ,UAAU,KAAK;CAC3C,CAAC;CAGH,MAAM,OAAO,IAAI,IACf,QAAQ,KAAK,MAAM,CAAC,EAAE,OAAO;EAAE,QAAQ,EAAE;EAAS,YAAY,EAAE;CAAY,CAAC,CAAC,CAChF;CACA,MAAM,UAAU;EACd,WAAW,OAAe,KAAK,IAAI,EAAE,CAAC,EAAE,UAAU;EAClD,eAAe,OAAe,KAAK,IAAI,EAAE,CAAC,EAAE,cAAc;EAC1D,0BAA0B,KAAK;CACjC;CACA,MAAM,EAAE,SAAS,MAAM,gBAAgB,SAAS,IAAI,mBAAmB,CAAC;CACxE,OAAO,eAAe,MAAM,SAAS,MAAM;AAC7C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACjBA,SAAgB,4BACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAElD,IAAI,aAAa,WAAW,GAC1B,OAAO,aAAa;CAGtB,MAAM,UAAoB,CAAC;CAC3B,MAAM,kBAA4B,CAAC;CACnC,IAAI,OAAO;CACX,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,gEAAgE,EAAE,MAAM,EAC1E;EAEF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,MAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;EAC7C,QAAQ,KAAK,CAAC;EACd,gBAAgB,KAAK,IAAI,CAAC;EAC1B,IAAI,IAAI,MAAM,OAAO;CACvB;CACA,MAAM,IAAI,QAAQ;CAClB,MAAM,QAAQ,gBAAgB,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3D,MAAM,WAAW,gBAAgB,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;CAC9F,MAAM,OAAO,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC;CAC9C,MAAM,QAAQ,QAAQ,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;CACnD,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAE9C,OAAO;EACL;EACA,eAAe,KAAK,KAAK,WAAW,CAAC;EACrC,qBAAqB;EACrB;EACA,qBAAqB;CACvB;AACF;;;;;;AAOA,SAAgB,kCACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAClD,IAAI,aAAa,WAAW,GAAG,OAAO,aAAa;CAEnD,MAAM,UAAoB,CAAC;CAC3B,MAAM,UAAoB,CAAC;CAC3B,IAAI,OAAO;CACX,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,sEAAsE,EAAE,MAAM,EAChF;EAEF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,QAAQ,KAAK,CAAC;EACd,QAAQ,KAAK,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI,CAAC;EACjD,IAAI,IAAI,MAAM,OAAO;CACvB;CACA,MAAM,OAAO,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC;CAC9C,MAAM,QAAQ,QAAQ,QAAQ,GAAG,GAAG,MAAM,IAAI,IAAI,QAAQ,IAAK,CAAC;CAChE,MAAM,QAAQ,SAAS,IAAI,IAAI,QAAQ;CACvC,MAAM,QAAQ,QAAQ,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;CACnD,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAG9C,MAAM,WADM,QAAQ,KAAK,GAAG,MAAM,KAAK,QAAQ,KAAM,MAClC,CAAC,CAAC,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,OAAO,IAAI;CAC7E,OAAO;EACL;EACA,eAAe,KAAK,KAAK,QAAQ;EACjC,qBAAqB;EACrB,GAAG,aAAa;EAChB,qBAAqB;CACvB;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,SAAgB,aACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAClD,IAAI,aAAa,WAAW,GAC1B,OAAO;EACL,GAAG,aAAa;EAChB,oBAAoB;GAAE,IAAI;GAAG,aAAa;EAAE;CAC9C;CAGF,MAAM,gBAA0B,CAAC;CACjC,MAAM,qBAAkD;EACtD,IAAI;EACJ,aAAa;CACf;CACA,IAAI,OAAO;CACX,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBAAgB,iDAAiD,EAAE,MAAM,EAAE;EAEvF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,MAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;EAC7C,MAAM,gBAAgB,EAAE;EACxB,MAAM,gBAAgB,EAAE;EACxB,MAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB,KAAA;EAClE,MAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB,KAAA;EAClE,IAAI,kBAAkB,eACpB,MAAM,IAAI,gBACR,4EAA4E,EAAE,MAAM,EACtF;EAGF,IAAI,iBAAiB,eAAe;GAClC,IAAI,CAAC,OAAO,SAAS,aAAa,KAAK,CAAC,OAAO,SAAS,aAAa,GACnE,MAAM,IAAI,gBACR,iEAAiE,EAAE,MAAM,EAC3E;GAEF,MAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;GAC3D,MAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;GAC3D,cAAc,KAAK,aAAa,KAAK,IAAI,WAAW;GACpD,mBAAmB,MAAM;EAC3B,OAAO;GACL,cAAc,KAAK,IAAI,CAAC;GACxB,mBAAmB,eAAe;EACpC;EACA,IAAI,IAAI,MAAM,OAAO;EACrB,QAAQ;EACR,SAAS,IAAI;CACf;CACA,MAAM,IAAI,cAAc;CACxB,MAAM,QAAQ,cAAc,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACzD,MAAM,WAAW,cAAc,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;CAC5F,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAC9C,OAAO;EACL;EACA,eAAe,KAAK,KAAK,WAAW,CAAC;EACrC,qBAAqB;EACrB;EACA,qBAAqB;EACrB;CACF;AACF;;;;;;AAOA,SAAgB,qBACd,cACA,OAAyB,CAAC,GACmD;CAC7E,OAAO;EACL,KAAK,4BAA4B,cAAc,IAAI;EACnD,OAAO,kCAAkC,cAAc,IAAI;EAC3D,IAAI,aAAa,cAAc,IAAI;CACrC;AACF;AAIA,SAAS,eAAkC;CACzC,OAAO;EAAE,OAAO;EAAG,eAAe;EAAG,qBAAqB;EAAG,GAAG;EAAG,qBAAqB;CAAE;AAC5F;AAEA,SAAS,MAAM,GAAW,IAAY,IAAoB;CACxD,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,CAAC,CAAC;AACrC;;;;;;;AChQA,IAAa,+BAAb,MAAgE;CAC9D;CACA,aAA4D;CAE5D,YAAY,MAA2C;EACrD,KAAK,OAAO;CACd;CAEA,MAAM,gBAAgB,MAA2C;EAC/D,MAAM,YAAY,KAAK,KAAK,oBAAoB;EAChD,MAAM,WAA0B,CAAC;EAIjC,MAAM,cAAc,KAAK,QAAQ,MAAM;GACrC,MAAM,QAAQ,aAAa,CAAC;GAC5B,OAAO,OAAO,UAAU,YAAY,QAAQ;EAC9C,CAAC;EACD,IAAI,YAAY,WAAW,GAAG,OAAO;EAIrC,MAAM,0BAAU,IAAI,IAAyB;EAC7C,KAAK,MAAM,KAAK,aAAa;GAC3B,MAAM,MAAM,QAAQ,IAAI,EAAE,WAAW,KAAK,CAAC;GAC3C,IAAI,KAAK,CAAC;GACV,QAAQ,IAAI,EAAE,aAAa,GAAG;EAChC;EAEA,KAAK,MAAM,CAAC,aAAa,UAAU,QAAQ,QAAQ,GAAG;GACpD,MAAM,YACJ,MAAM,QAAQ,GAAG,MAAM;IACrB,MAAM,QAAQ,aAAa,CAAC;IAC5B,IAAI,UAAU,KAAA,GACZ,MAAM,IAAI,MAAM,eAAe,EAAE,MAAM,gCAAgC;IAEzE,OAAO,IAAI;GACb,GAAG,CAAC,IAAI,MAAM;GAChB,SAAS,KAAK;IACZ,MAAM,aAAa;IACnB,aAAa,GAAG,YAAY,YAAY,UAAU,MAAM,MAAM,OAAO,gBAAgB,UAAU,QAAQ,CAAC,EAAE;IAC1G,UAAU;KACR,QAAQ,MAAM,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,MAAM,EAAE,KAAK;KAC5C,SAAS,MAAM;IACjB;GACF,CAAC;EACH;EACA,OAAO;CACT;CAEA,MAAM,cAAc,UAAoD;EACtE,IAAI,SAAS,WAAW,GAAG,OAAO,CAAC;EAInC,IAAI,KAAK,eAAe,MACtB,OAAO,CACL;GACE,MAAM;GACN,SAAS,EAAE,WAAW,mCAAmC;GACzD,WACE;EACJ,CACF;EAGF,MAAM,sBAAsB,KAAK,KAAK,uBAAuB;EAC7D,MAAM,UAA4B,CAAC;EAEnC,KAAK,MAAM,WAAW,KAAK,WAAW,QAAQ;GAC5C,IAAI,QAAQ,YAAY,gBAAgB;GACxC,IAAI,KAAK,IAAI,QAAQ,QAAQ,KAAK,qBAAqB;GACvD,QAAQ,KAAK;IACX,MAAM;IACN,SAAS;KACP,QAAQ,QAAQ;KAChB,QAAQ;KACR,UAAU,QAAQ;KAClB,aAAa,QAAQ;IACvB;IACA,WAAW,gCAAgC,QAAQ,SAAS,QAAQ,CAAC,EAAE,MAAM,QAAQ,YAAY;IACjG,eAAe,CAAC,KAAK,IAAI,GAAG,MAAO,KAAK,IAAI,QAAQ,QAAQ,CAAC;GAC/D,CAAC;EACH;EACA,KAAK,MAAM,WAAW,KAAK,WAAW,OAAO,MAAM,GAAG,CAAC,GAAG;GACxD,IAAI,QAAQ,YAAY,gBAAgB;GACxC,QAAQ,KAAK;IACX,MAAM;IACN,SAAS;KACP,QAAQ,QAAQ;KAChB,QAAQ;KACR,UAAU,QAAQ;KAClB,aAAa,QAAQ;IACvB;IACA,WAAW,gCAAgC,QAAQ,SAAS,QAAQ,CAAC,EAAE,MAAM,QAAQ,YAAY;IACjG,eAAe,KAAK,IAAI,GAAG,KAAK,IAAI,QAAQ,QAAQ,IAAI,EAAG,IAAI;GACjE,CAAC;EACH;EACA,OAAO;CACT;CAEA,MAAM,YAAY,SAA2B,UAAmD;EAG9F,OAAO;GACL,GAAG;GACH,SAAS,CAAC,GAAG,SAAS,SAAS,GAAG,OAAO;EAC3C;CACF;CAEA,MAAM,eAAe,MAAiD;EAqCpE,OAAO;GACL;GACA,MAAM,CAAC;GACP,cAAc;IAlCd,SAAS;IACT,aAAa,KAAK;IAClB,YAAY,KAAK;IACjB,UAAU;KACR,gBAAgB;KAChB,uBAAuB;KACvB,sBAAsB;KACtB,mBAAmB;KACnB,gBAAgB;KAChB,eAAe;KACf,UAAU;KACV,cAAc;KACd,SAAS;KACT,aAAa;KACb,aAAa;KACb,aAAa;KACb,cAAc;KACd,YAAY;KACZ,oBAAoB;KACpB,qBAAqB;KACrB,oBAAoB;KACpB,mBAAmB;KAGnB,iBAAiB,cAAc;KAC/B,gBAAgB,cAAc;IAChC;IACA,QACE;IACF,eAAe;GAKO;EACxB;CACF;;;;;;CAOA,MAAM,iBAAiB,MAA4D;EACjF,MAAM,SAAS,MAAM,yBAAyB;GAC5C;GACA,UAAU,KAAK,KAAK;GACpB,gBAAgB,KAAK,KAAK;GAC1B,SAAS,KAAK,KAAK;EACrB,CAAC;EACD,IAAI,KAAK,KAAK,UAAU,MAAM,KAAK,KAAK,SAAS,MAAM;EACvD,KAAK,aAAa;EAClB,OAAO;CACT;;;;;;CAOA,UAAU,QAA8C;EACtD,KAAK,aAAa;CACpB;CAEA,gBAAuD;EACrD,OAAO,KAAK;CACd;AACF;;AAGA,SAAS,gBAA+B;CACtC,OAAO;EAAE,OAAO;EAAG,UAAU;EAAG,eAAe;EAAG,eAAe;EAAG,cAAc;EAAG,UAAU;CAAE;AACnG;;;;ACrHA,MAAM,gBAA8B;;;;;;;;;;;;;;;;;;;AAmCpC,SAAgB,mBACd,OACA,OAAkC,CAAC,GACP;CAC5B,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,iBAAiB,KAAK;CAC5B,IAAI,mBAAmB,aAAa,KAAK,6BAA6B,MACpE,MAAM,IAAI,MAAM,+EAA6E;CAE/F,IAAI,mBAAmB,SAAS,mBAAmB,UACjD,MAAM,IAAI,MACR,8BAA8B,eAAe,0CAC/C;CAEF,MAAM,aAAa,oBAAoB,OAAO,IAAI;CAElD,OAAO;EAAE,GADM,eAAe,WAAW,MAAM,UAAU,SACxC;EAAG,yBAAyB,WAAW;CAAmB;AAC7E;AAOA,SAAS,oBACP,OACA,MACiB;CACjB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,OAA2B,CAAC;CAClC,IAAI,qBAAqB;CACzB,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,KAAK,UAAU,OAAO;EAC/B,IAAI,CAAC,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,UAAU,MACpF;EAEF,MAAM,QAAQ,oBAAoB,IAAI;EACtC,IAAI,UAAU,MAAM;EACpB,MAAM,cAAc,KAAK;EACzB,IAAI,gBAAgB,QAAQ,gBAAgB,KAAA,KAAa,YAAY,WAAW,GAAG;GACjF;GACA;EACF;EACA,KAAK,KAAK;GACR,YAAY,KAAK,KAAK;GACtB,OAAO,KAAK;GACZ;GACA,MAAM,KAAK,KAAK;GAChB;GAIA,YAAY,KAAK,OAAO,eAAe;GACvC,YAAY,KAAK,OAAO,eAAe;GACvC,OAAO,KAAK,OAAO,SAAS;EAC9B,CAAC;CACH;CACA,OAAO;EAAE;EAAM;CAAmB;AACpC;AAEA,SAAS,eACP,eACA,UACA,WAC6D;CAC7D,MAAM,QAA4B,CAAC;CACnC,IAAI,mBAAmB;CACvB,IAAI,iBAAiB;CACrB,IAAI,iBAAiB;CAErB,IAAI,aAAa,+BAA+B;EAE9C,MAAM,yBAAS,IAAI,IAAgC;EACnD,KAAK,MAAM,KAAK,eAAe;GAC7B,MAAM,MAAM,GAAG,EAAE,WAAW,IAAI,EAAE;GAClC,MAAM,MAAM,OAAO,IAAI,GAAG,KAAK,CAAC;GAChC,IAAI,KAAK,CAAC;GACV,OAAO,IAAI,KAAK,GAAG;EACrB;EAEA,KAAK,MAAM,WAAW,OAAO,OAAO,GAAG;GACrC;GACA,IAAI,QAAQ,SAAS,GAAG;IACtB;IACA;GACF;GACA,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAClC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAAK;IAC3C,MAAM,IAAI,QAAQ;IAClB,MAAM,IAAI,QAAQ;IAClB,IAAI,EAAE,gBAAgB,EAAE,aAAa;IACrC,MAAM,SAAS,SAAS,GAAG,GAAG,EAAE,YAAY,SAAS;IACrD,IAAI,OAAO,SAAS,SAAS,MAAM,KAAK,OAAO,IAAI;SAC9C;GACP;EAEJ;CACF,OAAO,IAAI,aAAa,sBAAsB;EAE5C,MAAM,oCAAoB,IAAI,IAG5B;EACF,KAAK,MAAM,KAAK,eAAe;GAC7B,IAAI,cAAc,kBAAkB,IAAI,EAAE,UAAU;GACpD,IAAI,CAAC,aAAa;IAChB,8BAAc,IAAI,IAAI;IACtB,kBAAkB,IAAI,EAAE,YAAY,WAAW;GACjD;GACA,MAAM,MAAM,YAAY,IAAI,EAAE,WAAW;GACzC,IAAI,KAAK;IACP,IAAI,OAAO,EAAE;IACb,IAAI;GACN,OAAO,YAAY,IAAI,EAAE,aAAa;IAAE,OAAO;IAAG,KAAK,EAAE;IAAO,GAAG;GAAE,CAAC;EACxE;EACA,KAAK,MAAM,CAAC,KAAK,eAAe,kBAAkB,QAAQ,GAAG;GAC3D;GACA,MAAM,MAAM,CAAC,GAAG,WAAW,OAAO,CAAC,CAAC,CAAC,KAAK,SAAS;IACjD,GAAG,IAAI;IACP,OAAO,IAAI,MAAM,IAAI;GACvB,EAAE;GACF,IAAI,IAAI,SAAS,GAAG;IAClB;IACA;GACF;GACA,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAC9B,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;IACvC,MAAM,SAAS,SAAS,IAAI,IAAK,IAAI,IAAK,KAAK,SAAS;IACxD,IAAI,OAAO,SAAS,SAAS,MAAM,KAAK,OAAO,IAAI;SAC9C;GACP;EAEJ;CACF,OAAO;EAEL,MAAM,6BAAa,IAAI,IAAgC;EACvD,KAAK,MAAM,KAAK,eAAe;GAC7B,MAAM,MAAM,WAAW,IAAI,EAAE,UAAU,KAAK,CAAC;GAC7C,IAAI,KAAK,CAAC;GACV,WAAW,IAAI,EAAE,YAAY,GAAG;EAClC;EACA,KAAK,MAAM,CAAC,KAAK,QAAQ,WAAW,QAAQ,GAAG;GAC7C;GACA,IAAI,IAAI,SAAS,GAAG;IAClB;IACA;GACF;GACA,MAAM,SAAS,CAAC,GAAG,GAAG,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;GACxD,MAAM,MAAM,OAAO,OAAO,SAAS;GACnC,MAAM,MAAM,OAAO;GACnB,IAAI,IAAI,gBAAgB,IAAI,aAAa;IACvC;IACA;GACF;GACA,MAAM,SAAS,SAAS,KAAK,KAAK,KAAK,SAAS;GAChD,IAAI,OAAO,SAAS,SAAS,MAAM,KAAK,OAAO,IAAI;QAC9C;EACP;CACF;CAEA,OAAO;EAAE;EAAO;EAAgB;EAAkB;EAAgB;CAAS;AAC7E;AAEA,MAAM,sBAAsB,MAA2C,CACrE,EAAE,aACF,EAAE,aACJ;AAEA,MAAM,0BAAkD;CACtD,UAAU;CACV,aAAa;CACb,SACE;AACJ;AAEA,MAAM,gCAAwD;CAC5D,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;;;;;;;;;;;AAmBA,eAAsB,YACpB,SACA,SACA,SACsE;CACtE,MAAM,WAAW,yBACf,SACA,oBACA,SACA,uBACF;CACA,MAAM,MAAmE,CAAC;CAC1E,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,CAAC,cAAc,gBAAgB,QAAQ,YAAY,MAAM,QAAQ,IAAI;GACzE,QAAQ,QAAQ,QAAQ,SAAS,EAAE,WAAW,CAAC;GAC/C,QAAQ,QAAQ,QAAQ,SAAS,EAAE,aAAa,CAAC;GACjD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,WAAW,CAAC;GACnD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,aAAa,CAAC;EACvD,CAAC;EACD,IAAI,iBAAiB,gBACnB,MAAM,IAAI,MACR,4BAA4B,EAAE,YAAY,KAAK,EAAE,cAAc,gCACjE;EAEF,IAAI,KAAK;GAAE,QAAQ;GAAc;GAAQ;EAAS,CAAC;CACrD;CACA,OAAO;AACT;;;;;;;;;;AAWA,SAAgB,kBACd,SACA,SAC2F;CAC3F,OAAO,yBACL,SACA,oBACA,SACA,6BACF,CAAC,CAAC,KAAK,OAAO;EACZ,YAAY,EAAE;EACd,aAAa,EAAE;EACf,eAAe,EAAE;EACjB,QAAQ,EAAE;CACZ,EAAE;AACJ;AAIA,SAAS,SACP,GACA,GACA,YACA,WACgE;CAEhE,IADe,KAAK,IAAI,EAAE,QAAQ,EAAE,KAC3B,IAAI,WAAW,OAAO,EAAE,MAAM,SAAS;CAChD,MAAM,CAAC,QAAQ,YAAY,EAAE,QAAQ,EAAE,QAAQ,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC;CAC7D,MAAM,OAAO,OAAO,SAAS,QAAQ,OAAO,SAAS,SAAS,OAAO,OAAO,OAAO,KAAA;CACnF,OAAO;EACL,MAAM;EACN,MAAM;GACJ;GACA,aAAa,OAAO;GACpB,eAAe,SAAS;GACxB,iBAAiB,OAAO;GACxB,mBAAmB,SAAS;GAC5B,aAAa,OAAO,QAAQ,SAAS;GACrC,QAAQ;IAAE,QAAQ,OAAO;IAAO,UAAU,SAAS;GAAM;GACzD;GACA,MAAM;IACJ,kBAAkB,OAAO;IACzB,oBAAoB,SAAS;IAC7B,kBAAkB,OAAO;IACzB,oBAAoB,SAAS;IAC7B,aAAa,OAAO;IACpB,eAAe,SAAS;GAC1B;EACF;CACF;AACF;;;ACtXA,eAAsB,mBACpB,OACA,OACA,MACuB;CAEvB,MAAM,UAAU,CAAC,GAAG,MADA,MAAM,MAAM,EAAE,MAAM,CAAC,CAChB,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACnE,MAAM,MAAoB,CAAC;CAC3B,IAAI,MAAM;CACV,KAAK,MAAM,QAAQ,SAAS;EAC1B,IAAI,KAAK,aAAa,CAAC,KAAK,UAAU,IAAI,GAAG;EAC7C,IAAI,SAAmD;EACvD,KAAK,MAAM,KAAK,KAAK,SAAS;GAC5B,IAAI,CAAC,EAAE,UAAU,SAAS,KAAK,IAAI,GAAG;GACtC,MAAM,IAAI,MAAM,EAAE,MAAM,IAAI;GAC5B,IAAI,GAAG;IACL,SAAS;IACT;GACF;EACF;EACA,IAAI,CAAC,QAAQ;EACb,IAAI,KAAK;GACP,QAAQ,KAAK;GACb;GACA,WAAW;GACX,MAAM,KAAK;GACX,MAAM,KAAK;GACX,QAAQ,OAAO;GACf,aAAa,OAAO;GACpB,WAAW,OAAO;GAClB,QAAQ,OAAO;EACjB,CAAC;CACH;CACA,OAAO;AACT;AAeA,SAAgB,yBAAyB,aAA+C;CACtF,IAAI,YAAY,WAAW,GACzB,OAAO;EACL,OAAO;EACP,YAAY;EACZ,YAAY;EACZ,mBAAmB;EACnB,iBAAiB;EACjB,gBAAgB;EAChB,gBAAgB;CAClB;CAEF,MAAM,QAAQ,YAAY,EAAE,CAAE;CAC9B,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,IAAI,WAAW;CACf,IAAI,aAAa;CACjB,IAAI,WAA0B;CAC9B,IAAI,OAAO,YAAY,EAAE,CAAE;CAC3B,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,QAAQ,KAAK;EAC3C,MAAM,IAAI,YAAY;EACtB,MAAM,IAAI,EAAE,UAAU;EACtB,QAAQ;EACR,SAAS,IAAI,EAAE;EACf,IAAI,EAAE,SAAS,IAAK;EACpB,IAAI,IAAI,GAAG;GACT,MAAM,QAAQ,EAAE,SAAS;GACzB,IAAI,QAAQ,YAAY;IACtB,aAAa;IACb,WAAW;GACb;GACA,OAAO,EAAE;EACX,OACE,OAAO,EAAE;CAEb;CACA,OAAO;EACL;EACA,YAAY,YAAY;EACxB,YAAY,SAAS,IAAI,IAAI,QAAQ;EACrC,mBAAmB;EACnB,iBAAiB,WAAW,YAAY;EACxC,gBAAgB;EAChB,gBAAgB;CAClB;AACF;;;;;;;;;;;;;;;;AAgCA,SAAgB,iBACd,kBACA,OAAyD,CAAC,GACrC;CACrB,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,YAAY,KAAK,mBAAmB;CAC1C,MAAM,OAAO,CAAC,GAAG,iBAAiB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,YAAY;EAAE;EAAO;CAAM,EAAE;CACvF,MAAM,UAA+B,CAAC;CAEtC,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAC/B,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACxC,MAAM,IAAI,KAAK;EACf,MAAM,IAAI,KAAK;EACf,MAAM,SAAS,KAAK,IAAI,EAAE,MAAM,QAAQ,EAAE,MAAM,MAAM;EACtD,IAAI,SAAS,YAAY,GAAG;EAO5B,IAAI,gBAAgB;EACpB,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,KAAK;GAC/B,MAAM,KAAK,EAAE,MAAM;GACnB,MAAM,KAAK,EAAE,MAAM;GACnB,MAAM,uBAAuB,GAAG,SAAS,GAAG,QAAQ,GAAG,SAAS,GAAG;GACnE,MAAM,YAAY,KAAK,IAAI,GAAG,SAAS,GAAG,MAAM;GAChD,IAAI,wBAAwB,aAAa,WAAW;IAClD,gBAAgB;IAChB;GACF;EACF;EACA,IAAI,gBAAgB,GAAG;EACvB,IAAI,gBAAgB,WAAW;EAE/B,MAAM,QAAQ,EAAE,MAAM;EACtB,MAAM,QAAQ,EAAE,MAAM;EAEtB,IADe,KAAK,IAAI,MAAM,SAAS,MAAM,MACpC,IAAI,WAAW;EAExB,MAAM,SAAS,MAAM,SAAS,MAAM,SAAS,QAAQ;EACrD,MAAM,WAAW,MAAM,SAAS,MAAM,SAAS,QAAQ;EACvD,MAAM,YAAY,MAAM,SAAS,MAAM,SAAS,EAAE,QAAQ,EAAE;EAC5D,MAAM,cAAc,MAAM,SAAS,MAAM,SAAS,EAAE,QAAQ,EAAE;EAC9D,QAAQ,KAAK;GACX,aAAa;GACb,iBAAiB,gBAAgB;GACjC,cAAc,OAAO;GACrB,cAAc,OAAO;GACrB,gBAAgB,SAAS;GACzB,gBAAgB,SAAS;GACzB,eAAe;GACf,aAAa,OAAO,SAAS,SAAS;EACxC,CAAC;CACH;CAEF,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;AC9GA,eAAsB,cAAiB,MAA0D;CAC/F,MAAM,WAAW,KAAK,YAAY;CAGlC,MAAM,WAAW,MAAM,gBAAgB;EAAE,GAAG;EAAM;CAAS,CAAC;CAG5D,MAAM,gBAAgB,oCACpB,SAAS,MACT,KAAK,oBAAoB,CAAC,CAC5B;CAIA,MAAM,EAAE,MAAM,iBAAiB,MAAM,gBADlB,SAAS,KAAK,QAAQ,QAAQ,aAAa,GAAG,MAAM,KAAA,CACT,GAAG,IAAI,mBAAmB,CAAC;CACzF,MAAM,cAAc,mBAAmB,cAAc;EACnD,GAAG,KAAK;EACR,UAAU,KAAK,aAAa,YAAY;EACxC,WAAW,KAAK,aAAa,aAAa;EAC1C,OAAO,KAAK,aAAa,SAAS;CACpC,CAAC;CAGD,IAAI,oBAAqD;CACzD,IAAI,gBAAuC,CAAC;CAC5C,MAAM,mBAAmB,KAAK,YAAY,oBAAoB;CAC9D,IAAI,EAAE,OAAO,SAAS,gBAAgB,KAAK,oBAAoB,KAAK,oBAAoB,IACtF,MAAM,IAAI,MACR,uFAAuF,kBACzF;CAEF,IAAI,KAAK,QAAQ,YAAY;EAC3B,MAAM,aAAa,KAAK,OAAO;EAC/B,MAAM,SAAS,yBAAyB,SAAS,MAAM,SAAS,YAAY,UAAU;EACtF,gBAAgB,OAAO,KAAK,MAAM,EAAE,QAAQ;EAK5C,IADgB,OAAO,OAAO,MAAM,EAAE,SAAS,YAAY,gBACjD,KAAK,OAAO,MAAM,MAAM,EAAE,OAAO,SAAS,CAAC,GACnD,oBAAoB,iCAAiC;GACnD,aAAa,OAAO,KAAK,EAAE,aAAa,cAAc;IAAE;IAAa;GAAO,EAAE;GAC9E,OAAO,KAAK,YAAY;GACxB,OAAO,KAAK,YAAY;GACxB,MAAM,KAAK,YAAY,QAAQ,KAAK,QAAQ;EAC9C,CAAC;CAEL;CAGA,MAAM,gBAAgB,oBAAoB;EACxC,MAAM,SAAS;EACf,yBAAyB,KAAK;CAChC,CAAC;CAGD,IAAI,qBAA4D;CAChE,IAAI,KAAK,gBAAgB,KAAK,kBAAkB,KAAK,eAAe,SAAS,GAC3E,qBAAqB,MAAM,yBAAyB;EAClD,MAAM,SAAS;EACf,UAAU,KAAK;EACf,gBAAgB,KAAK;CACvB,CAAC;CAIH,MAAM,cAA+C,CAAC;CACtD,IAAI,KAAK,eAAe,KACtB,YAAY,MAAM,MAAM,UAAU,YAAY,OAAO,KAAK,cAAc,KAAK,EAC3E,OAAO,aACT,CAAC;CAEH,IAAI,KAAK,eAAe,MACtB,YAAY,OAAO,MAAM,WAAW,cAAc,KAAK,cAAc,IAAI;CAE3E,IAAI,KAAK,eAAe,KACtB,YAAY,MAAM,MAAM,UAAU,cAAc,KAAK,cAAc,GAAG;CAGxE,MAAM,UAAU,aAAa;EAC3B;EACA;EACA;EACA;EACA;EACA;CACF,CAAC;CAED,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,MAAM;CACR;AACF;;;;;;;;;;;;;;;;;AAoBA,SAAS,yBACP,MACA,YACA,YACiF;CACjF,MAAM,WAAW,MAA4C,GAAG,EAAE,WAAW,IAAI,EAAE;CAMnF,MAAM,oCAAoB,IAAI,IAAyB;CACvD,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,MAAM,kBAAkB,IAAI,EAAE,SAAS;EAC3C,IAAI,CAAC,KAAK;GACR,sBAAM,IAAI,IAAY;GACtB,kBAAkB,IAAI,EAAE,WAAW,GAAG;EACxC;EACA,IAAI,IAAI,QAAQ,CAAC,CAAC;CACpB;CAEA,MAAM,kBAAkB,IAAI,IAAY,kBAAkB,IAAI,UAAU,KAAK,CAAC,CAAC;CAC/E,MAAM,kCAAkB,IAAI,IAAoB;CAChD,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,EAAE,gBAAgB,YAAY;EAClC,MAAM,MAAM,QAAQ,CAAC;EACrB,gBAAgB,IAAI,GAAG;EAKvB,MAAM,QAAQ,aAAa,CAAC;EAC5B,IAAI,UAAU,KAAA,GAAW;EACzB,gBAAgB,IAAI,KAAK,KAAK;CAChC;CACA,MAAM,mCAAmB,IAAI,IAAyB;CACtD,MAAM,mCAAmB,IAAI,IAAiC;CAC9D,KAAK,MAAM,CAAC,WAAW,UAAU,mBAAmB;EAClD,IAAI,cAAc,YAAY;EAC9B,iBAAiB,IAAI,WAAW,IAAI,IAAI,KAAK,CAAC;CAChD;CACA,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,EAAE,gBAAgB,YAAY;EAClC,MAAM,MAAM,QAAQ,CAAC;EACrB,IAAI,QAAQ,iBAAiB,IAAI,EAAE,WAAW;EAC9C,IAAI,CAAC,OAAO;GACV,wBAAQ,IAAI,IAAY;GACxB,iBAAiB,IAAI,EAAE,aAAa,KAAK;EAC3C;EACA,MAAM,IAAI,GAAG;EACb,MAAM,QAAQ,aAAa,CAAC;EAC5B,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,SAAS,iBAAiB,IAAI,EAAE,WAAW;EAC/C,IAAI,CAAC,QAAQ;GACX,yBAAS,IAAI,IAAoB;GACjC,iBAAiB,IAAI,EAAE,aAAa,MAAM;EAC5C;EACA,OAAO,IAAI,KAAK,KAAK;CACvB;CAEA,OAAO,CAAC,GAAG,iBAAiB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,aAAa,oBAAoB;EAC5E,MAAM,SAAS,iBAAiB,IAAI,WAAW,qBAAK,IAAI,IAAoB;EAC5E,MAAM,SAAmB,CAAC;EAC1B,IAAI,oBAAoB;EACxB,IAAI,qBAAqB;EACzB,IAAI,YAAY;EAGhB,KAAK,MAAM,uBAAO,IAAI,IAAI,CAAC,GAAG,gBAAgB,GAAG,eAAe,CAAC,GAAG;GAClE,MAAM,cAAc,eAAe,IAAI,GAAG;GAC1C,MAAM,eAAe,gBAAgB,IAAI,GAAG;GAC5C,IAAI,CAAC,eAAe,CAAC,cAAc;IACjC,aAAa;IACb;GACF;GACA,MAAM,IAAI,OAAO,IAAI,GAAG;GACxB,MAAM,IAAI,gBAAgB,IAAI,GAAG;GACjC,IAAI,MAAM,KAAA,GAAW,qBAAqB;QACrC,IAAI,MAAM,KAAA,GAAW,sBAAsB;QAC3C,OAAO,KAAK,IAAI,CAAC;EACxB;EACA,MAAM,yBAAQ,IAAI,IAAI,CAAC,GAAG,gBAAgB,GAAG,eAAe,CAAC,EAAA,CAAE;EAC/D,OAAO;GACL;GACA;GACA,UAAU;IACR;IACA;IACA,UAAU,OAAO;IACjB;IACA;IACA;IACA,UAAU,UAAU,IAAI,IAAI,OAAO,SAAS;GAC9C;EACF;CACF,CAAC;AACH;AAEA,SAAS,aAAa,MAOX;CACT,MAAM,IAAI,KAAK;CACf,MAAM,QAAQ,CACZ,GAAG,EAAE,WAAW,IAAI,EAAE,KAAK,OAAO,qBAAqB,EAAE,WAAW,OAAO,uBAAuB,EAAE,oBAAoB,MAAM,GAAG,EAAE,EAAE,KACrI,gBAAgB,KAAK,YAAY,MAAM,OAAO,IAAI,KAAK,YAAY,SAAS,IAAI,KAAK,YAAY,iBAAiB,eACpH;CACA,IAAI,KAAK,mBACP,MAAM,KACJ,uBAAuB,KAAK,kBAAkB,eAAe,cAC1D,KAAK,kBAAkB,eAAe,cACnC,IAAI,KAAK,kBAAkB,eAAe,gBAC1C,GACR;CAIF,MAAM,YAAY,KAAK,cAAc,QAAQ,MAAM,EAAE,WAAW,EAAE,KAAK;CACvE,IAAI,UAAU,SAAS,GACrB,MAAM,KACJ,0BAA0B,UACvB,KAAK,MAAM,GAAG,EAAE,YAAY,GAAG,EAAE,SAAS,GAAG,EAAE,OAAO,CAAC,CACvD,KAAK,IAAI,IAAI,KAAK,oBAAoB,KAAK,kCAChD;CAEF,MAAM,KACJ,mBAAmB,KAAK,cAAc,QAAQ,IAAI,KAAK,cAAc,SAAS,OAAO,kBACvF;CACA,IAAI,KAAK,oBAAoB;EAC3B,MAAM,MAAM,KAAK,mBAAmB,OAAO;EAC3C,MAAM,KACJ,eAAe,KAAK,UAAU,OAAO,MAAM,KAAK,YAAY,EAAA,CAAG,QAAQ,CAAC,EAAE,IAAI,KAAK,WAAW,UAAU,EAC1G;CACF;CACA,OAAO,MAAM,KAAK,KAAK;AACzB;;;;;;;;;;;;;;;;;;;;;;;;AC/WA,SAAgB,qBACd,UACA,KACa;CACb,MAAM,WAAW,IAAI,YAAY;CACjC,MAAM,cAAc,IAAI,eAAe,SAAS;CAChD,OAAO,SAAS,MAAM,KAAK,SACzB,wBAAwB,MAAM;EAC5B,OAAO,KAAK;EACZ,cAAc,IAAI;EAClB;EACA,OAAO,IAAI;EACX,YAAY,IAAI;EAChB,YAAY,IAAI;EAChB,WAAW,IAAI;EACf;EACA,gBAAgB,IAAI;CACtB,CAAC,CACH;AACF;;;;;;;;AASA,SAAgB,8BACd,QACA,KACA,OAA2B,CAAC,GACjB;CACX,MAAM,WAAW,IAAI,YAAY;CACjC,MAAM,QAAQ,KAAK,SAAS,OAAO,IAAI,YAAY,GAAG,IAAI,aAAa,GAAG,OAAO;CAEjF,MAAM,YAD2B,OAAO,OAAO,KAAK,uBAE3B,KAAK,aAAa,OAAO,SAAS,IAAI,OAAO,YAAY,KAAA;CAClF,IAAI,sBAAsB;CAC1B,IAAI,kBAAkB;CACtB,IAAI,kBAAkB;CACtB,IAAI,oBAAoB;CACxB,IAAI,qBAAqB;CAEzB,MAAM,MAA8B;EAClC,YAAY,OAAO;EACnB,YAAY,OAAO;EACnB,aAAa,OAAO;EACpB,eAAe,OAAO;EACtB,aAAa,OAAO;EACpB,uBAAuB;CACzB;CACA,KAAK,MAAM,SAAS,OAAO,QAAQ;EACjC,IAAI,wBAAwB,KAAK,GAAG,IAAI,SAAS,MAAM,WAAW,MAAM;OACnE;EACL,IAAI,SAAS,MAAM,MAAM,UAAU,MAAM,WAAW,SAAS,IAAI;EACjE,IAAI,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;GAC1D,IAAI,MAAM,gBAAgB,SAAS;QAC9B;GACL,IAAI,MAAM,WAAW,SAAS;QACzB;EACP;EACA,IAAI,MAAM,aACH;QAAA,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,MAAM,WAAW,GACnD,IAAI,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,GAAG,IAAI,SAAS,MAAM,MAAM,GAAG,OAAO;EAAA;CAG1F;CAEA,IAAI,wBAAwB;CAC5B,IAAI,kBAAkB,GAAG,IAAI,oBAAoB;CACjD,IAAI,kBAAkB,GAAG,IAAI,oBAAoB;CACjD,IAAI,oBAAoB,GAAG,IAAI,sBAAsB;CACrD,IAAI,qBAAqB,GAAG,IAAI,uBAAuB;CACvD,IAAI,cAAc,KAAA,GAAW,IAAI,gBAAgB;CAEjD,MAAM,qBAAqB,OAAO,OAAO,MACtC,UAAU,MAAM,WAAW,UAAU,wBAAwB,KAAK,CACrE;CACA,MAAM,UAAgC,EAAE,IAAI;CAC5C,IAAI,cAAc,KAAA,GAChB,IAAI,aAAa,WAAW,QAAQ,eAAe;MAC9C,QAAQ,cAAc;CAG7B,OAAO;EACL;EACA,cAAc,IAAI;EAClB,aAAa,IAAI;EACjB,MAAM;EACN,OAAO,IAAI;EACX,YAAY,IAAI;EAChB,YAAY,IAAI;EAChB,WAAW,IAAI;EACf,QAAQ,OAAO;EACf,SAAS,IAAI,kBAAkB;EAC/B,gBACE,IAAI,mBAAmB,KAAA,IACnB;GAAE,MAAM;GAAc,KAAK;EAAK,IAChC;GAAE,MAAM;GAAa,KAAK,IAAI;EAAe;EACnD,YAAY;GAAE,OAAO;GAAG,QAAQ;EAAE;EAClC,iBAAiB;EACjB;EACA,GAAI,qBACA;GACE,cAAc;GACd,aAAa,SAAS,mBAAmB,MAAM;EACjD,IACA,CAAC;EACL;EACA,YAAY,IAAI;CAClB;AACF;AAEA,SAAS,wBACP,OACmE;CACnE,QAAQ,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,aAAa,MAAM,KAAK;AACzF;AAEA,SAAS,aAAa,OAAiC;CACrD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,KAAK,SAAS,KAAK,SAAS;AACvF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACvHA,MAAa,kBAAkB;;;AAI/B,MAAM,4BAA4B;;;AAIlC,MAAM,2BAA2B;;;;;AAMjC,MAAM,8BAA8B;AAEpC,MAAM,kBAAkB;;;;;;;;;;;;;;;;;;;;;;AAuBxB,MAAa,2BAA6C,WAAW;CACnE,MAAM,MAA8B,OAAO,SAAS,OAAO,CAAC;CAC5D,MAAM,aAAa,aAAa,IAAI,WAAW;CAC/C,MAAM,eAAe,aAAa,IAAI,aAAa;CACnD,MAAM,aAAc,OAAwB;CAC5C,OAAO;EAOL,OACE,aAAa,mBAAmB,QAAQ,SAAS,CAAC,KAClD,aAAa,mBAAmB,QAAQ,QAAQ,CAAC;EACnD,eAAe,OAAO,gBAAgB;EACtC,SAAS,aAAa,OAAO,MAAM;EACnC,eAAe,aAAa,OAAO,YAAY,MAAM;EACrD,YAAY,aAAa,IAAI,eAAe,KAAK,aAAa,IAAI,kBAAkB;EACpF,aAAa;EACb,qBAAqB,kBAAkB,YAAY,cAAc,OAAO,YAAY;EACpF,mBAAmB,OAAO,eAAe,WAAW,WAAW,SAAS;CAC1E;AACF;AAEA,SAAS,kBACP,YACA,cACA,cACe;CACf,IAAI,eAAe,MAAM,OAAO;CAChC,IAAI,eAAe,GAAG,OAAO;CAG7B,QADG,gBAAgB,KAAK,KAAM,iBAAiB,KAAA,KAAa,iBAAiB,YAC7D,gBAAgB;AAClC;AAEA,SAAS,aAAa,OAA+B;CACnD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ;AACvE;;;;;;;;;AAYA,SAAgB,aAAa,GAA2B,GAAmC;CACzF,MAAM,uBAAO,IAAI,IAAI,CAAC,GAAG,OAAO,KAAK,CAAC,GAAG,GAAG,OAAO,KAAK,CAAC,CAAC,CAAC;CAC3D,IAAI,KAAK,SAAS,GAChB,MAAM,IAAI,gBAAgB,yCAAyC;CAErE,IAAI,OAAO;CACX,IAAI,OAAO;CACX,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,EAAE,QAAQ;EACrB,MAAM,KAAK,EAAE,QAAQ;EACrB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,CAAC,OAAO,SAAS,EAAE,KAAK,KAAK,KAAK,KAAK,GACjE,MAAM,IAAI,gBAAgB,4DAA4D,IAAI,EAAE;EAE9F,QAAQ;EACR,QAAQ;CACV;CACA,IAAI,SAAS,KAAK,SAAS,GACzB,MAAM,IAAI,gBAAgB,oEAAoE;CAEhG,IAAI,aAAa;CACjB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,MAAM,EAAE,QAAQ,KAAK;EAC3B,MAAM,MAAM,EAAE,QAAQ,KAAK;EAC3B,MAAM,KAAK,KAAK,MAAM;EACtB,IAAI,KAAK,GAAG,cAAc,KAAM,KAAK,KAAK,KAAK,KAAK,CAAC;EACrD,IAAI,KAAK,GAAG,cAAc,KAAM,KAAK,KAAK,KAAK,KAAK,CAAC;CACvD;CAEA,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,UAAU,CAAC;AAC5C;;;;;;;AAQA,SAAgB,cAAc,QAAkB,cAAc,0BAAoC;CAChG,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,GAClD,MAAM,IAAI,gBACR,2DAA2D,aAC7D;CAEF,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,QAAkB,CAAC;CACzB,KAAK,IAAI,IAAI,GAAG,IAAI,aAAa,KAAK;EACpC,MAAM,MAAO,IAAI,eAAgB,OAAO,SAAS;EACjD,MAAM,KAAK,OAAO,KAAK,MAAM,GAAG;EAChC,MAAM,KAAK,OAAO,KAAK,KAAK,GAAG;EAC/B,MAAM,KAAK,MAAM,MAAM,KAAK,MAAM,GAAG,MAAM,KAAK,GAAG;CACrD;CACA,OAAO,CAAC,GAAG,IAAI,IAAI,KAAK,CAAC;AAC3B;;;AAIA,SAAgB,YAAY,OAAe,OAAyB;CAClE,IAAI,IAAI;CACR,OAAO,IAAI,MAAM,UAAU,SAAS,MAAM,IAAK;CAG/C,OAAO,IAFI,MAAM,IAAI,SAAS,OAAO,MAAM,IAAI,EAAG,EAEpC,GADH,MAAM,MAAM,SAAS,SAAS,OAAO,MAAM,EAAG,EACrC;AACtB;;;;;;;AAuDA,SAAgB,kBACd,WACA,YACA,OAA2B,CAAC,GACZ;CAChB,IAAI,UAAU,WAAW,GACvB,MAAM,IAAI,gBAAgB,gDAAgD;CAE5E,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,gBAAgB,iDAAiD;CAE7E,MAAM,UAAU,KAAK,YAAY;CACjC,MAAM,OAAO,KAAK,kBAAkB;CAEpC,MAAM,UAAU,UAAU,IAAI,OAAO;CACrC,MAAM,WAAW,WAAW,IAAI,OAAO;CAIvC,MAAM,eAAyB,CAAC;CAChC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,OAAO,CAAC,GAAG,SAAS,GAAG,QAAQ,GACxC,KAAK,MAAM,QAAQ,OAAO,KAAK,GAAG,GAChC,IAAI,CAAC,KAAK,IAAI,IAAI,GAAG;EACnB,KAAK,IAAI,IAAI;EACb,aAAa,KAAK,IAAI;CACxB;CAIJ,MAAM,eAAoC,CAAC;CAC3C,MAAM,mBAA6B,CAAC;CAEpC,KAAK,MAAM,WAAW,cAAc;EAClC,MAAM,UAAU,QAAQ,KAAK,MAAM,EAAE,YAAY,IAAI;EACrD,MAAM,WAAW,SAAS,KAAK,MAAM,EAAE,YAAY,IAAI;EACvD,MAAM,OAAO,QAAQ,QAAQ,MAAM,MAAM,IAAI,CAAC,CAAC;EAC/C,MAAM,QAAQ,SAAS,QAAQ,MAAM,MAAM,IAAI,CAAC,CAAC;EACjD,IAAI,OAAO,QAAQ,QAAQ,MAAM;GAC/B,iBAAiB,KAAK,OAAO;GAC7B;EACF;EACA,MAAM,EAAE,KAAK,SAAS,WAAW,SAAS,SAAS,QAAQ;EAC3D,aAAa,KAAK;GAChB;GACA,YAAY,aAAa,KAAK,IAAI;GAClC,WAAW,UAAU,KAAK,QAAQ,QAAQ,MAAM,SAAS,MAAM;GAC/D;GACA;EACF,CAAC;CACH;CAEA,IAAI,aAAa,WAAW,GAC1B,OAAO;EAAE;EAAc,UAAU;EAAY;EAAkB,SAAS;CAAoB;CAE9F,MAAM,WAAW,IAAI,aAAa,QAAQ,KAAK,MAAM,MAAM,EAAE,YAAY,CAAC,IAAI,aAAa;CAC3F,OAAO;EACL;EACA;EACA;EACA,SAAS,YAAY,8BAA8B,mBAAmB;CACxE;AACF;AAIA,SAAS,WACP,SACA,SACA,UAC+D;CAC/D,MAAM,wBAAQ,IAAI,IAAY;CAC9B,KAAK,MAAM,KAAK,CAAC,GAAG,SAAS,GAAG,QAAQ,GACtC,IAAI,MAAM,MAAM,MAAM,IAAI,OAAO,CAAC;CAEpC,IAAI,MAAM,OAAO,GACf,MAAM,IAAI,gBACR,+BAA+B,QAAQ,iFACzC;CAEF,IAAI;CACJ,IAAI,MAAM,IAAI,QAAQ,GAAG;EACvB,MAAM,QAAkB,CAAC;EACzB,KAAK,MAAM,KAAK,CAAC,GAAG,SAAS,GAAG,QAAQ,GACtC,IAAI,MAAM,MAAM,MAAM,KAAK,CAAW;EAExC,MAAM,QAAQ,cAAc,KAAK;EACjC,cAAc,MAAM,YAAY,GAAa,KAAK;CACpD,OACE,cAAc,MAAM;CAEtB,MAAM,SAAS,SAAiD;EAC9D,MAAM,OAA+B,CAAC;EACtC,KAAK,MAAM,KAAK,MAAM;GACpB,MAAM,MAAM,MAAM,OAAO,kBAAkB,WAAW,CAAC;GACvD,KAAK,QAAQ,KAAK,QAAQ,KAAK;EACjC;EACA,OAAO;CACT;CACA,OAAO;EAAE,KAAK,MAAM,OAAO;EAAG,MAAM,MAAM,QAAQ;CAAE;AACtD;AAEA,SAAS,UACP,KACA,UACA,MACA,WACgB;CAEhB,MAAM,SAAS,CADD,mBAAG,IAAI,IAAI,CAAC,GAAG,OAAO,KAAK,GAAG,GAAG,GAAG,OAAO,KAAK,IAAI,CAAC,CAAC,CAClD,CAAC,CAAC,KAAK,WAAW;EAClC;EACA,OAAO,IAAI,UAAU,KAAK;EAC1B,QAAQ,KAAK,UAAU,KAAK;CAC9B,EAAE;CACF,OAAO,MAAM,GAAG,MAAM;EACpB,MAAM,QAAQ,KAAK,IAAI,EAAE,OAAO,EAAE,KAAK,IAAI,KAAK,IAAI,EAAE,OAAO,EAAE,KAAK;EACpE,OAAO,UAAU,IAAI,QAAQ,EAAE,MAAM,cAAc,EAAE,KAAK;CAC5D,CAAC;CACD,OAAO,OAAO,MAAM,GAAG,eAAe;AACxC;;;;;;;;AA8BA,SAAgB,cACd,WACA,YACA,OAAwB,CAAC,GACT;CAChB,IAAI,UAAU,WAAW,GACvB,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,gBAAgB,6CAA6C;CAEzE,MAAM,YAAY,KAAK,iBAAiB;CACxC,MAAM,YAAY,KAAK,sBAAsB;CAC7C,MAAM,YAAY,SAAsB,SAAyB;EAC/D,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,SAAS;GAIvB,MAAM,QACJ,aAAa,mBAAmB,GAAG,SAAS,CAAC,KAC7C,aAAa,mBAAmB,GAAG,QAAQ,CAAC;GAC9C,IAAI,UAAU,MACZ,MAAM,IAAI,gBACR,kBAAkB,KAAK,QAAQ,EAAE,MAAM,+CACzC;GAEF,IAAI,SAAS,WAAW;EAC1B;EACA,OAAO,SAAS,QAAQ;CAC1B;CACA,MAAM,cAAc,SAAS,WAAW,WAAW;CACnD,MAAM,eAAe,SAAS,YAAY,YAAY;CACtD,MAAM,MAAM,cAAc;CAC1B,OAAO;EAAE;EAAa;EAAc;EAAK,UAAU,MAAM;CAAU;AACrE;;;;;;;;;;;;AC9WA,SAAgB,gBACd,UACA,OAA2E,CAAC,GAC3D;CACjB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,UAAU,KAAK,iBAAiB;CAKtC,MAAM,YAAY,KAAK,aAAa;CAEpC,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,UAAU;EACxB,WAAW,IAAI,EAAE,MAAM;EACvB,WAAW,IAAI,EAAE,KAAK;CACxB;CACA,MAAM,MAAM,CAAC,GAAG,UAAU,CAAC,CAAC,KAAK;CACjC,MAAM,MAAM,IAAI,IAAI,IAAI,KAAK,IAAI,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC;CAC/C,MAAM,IAAI,IAAI;CACd,IAAI,MAAM,GAAG,OAAO;EAAE,SAAS,CAAC;EAAG,YAAY;EAAG,YAAY;EAAG,WAAW;CAAK;CACjF,IAAI,MAAM,GACR,OAAO;EACL,SAAS,CAAC;GAAE,aAAa,IAAI;GAAK,UAAU;GAAG,aAAa;GAAG,GAAG;GAAG,MAAM;EAAE,CAAC;EAC9E,YAAY;EACZ,YAAY;EACZ,WAAW;CACb;CAKF,MAAM,IAAgB,MAAM,KAAK,EAAE,QAAQ,EAAE,SAAS,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;CAClF,MAAM,IAAgB,MAAM,KAAK,EAAE,QAAQ,EAAE,SAAS,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;CAClF,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,IAAI,IAAI,IAAI,EAAE,MAAM;EAC1B,MAAM,IAAI,IAAI,IAAI,EAAE,KAAK;EACzB,MAAM,IAAI,EAAE,UAAU;EACtB,IAAI,EAAE,MAAM;GACV,EAAE,EAAE,CAAE,MAAO,KAAM;GACnB,EAAE,EAAE,CAAE,MAAO,KAAM;EACrB,OACE,EAAE,EAAE,CAAE,MAAO;EAEf,EAAE,EAAE,CAAE,MAAO;EACb,EAAE,EAAE,CAAE,MAAO;CACf;CAGA,MAAM,YAAY,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,UAAU,MAAO,EAAE,EAAE,CAAE;EACnD,UAAU,MAAO;CACnB;CACA,MAAM,aAAa,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC;CAC9C,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,WAAW,MAAO,EAAE,EAAE,CAAE;CAItD,IAAI,QAAQ,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC;CACvC,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,OAAO,OAAO,SAAS,QAAQ;EAC7B,MAAM,WAAW,IAAI,MAAc,CAAC;EACpC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,IAAI,QAAQ;GACZ,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;IAC1B,IAAI,MAAM,GAAG;IACb,IAAI,EAAE,EAAE,CAAE,OAAQ,GAAG;IACrB,SAAS,EAAE,EAAE,CAAE,MAAO,MAAM,KAAM,MAAM;GAC1C;GACA,SAAS,KAAK,UAAU,IAAI,MAAM,KAAM,UAAU,KAAM;EAC1D;EAEA,IAAI,SAAS;EACb,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,UAAU,KAAK,IAAI,KAAK,IAAI,QAAQ,SAAS,EAAG,CAAC;EAC7E,MAAM,OAAO,KAAK,IAAI,SAAS,CAAC;EAChC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,SAAS,KAAK,SAAS,KAAM;EAEzD,QAAQ;EACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,IAAI,KAAK,IAAI,SAAS,KAAM,MAAM,EAAG,IAAI,KAAK,IAAI,OAAO,MAAM,EAAG;GACxE,IAAI,IAAI,OAAO,QAAQ;EACzB;EACA,QAAQ;EACR,IAAI,QAAQ,KAAK;CACnB;CAEA,MAAM,SAAS,KAAK,IAAI,GAAG,MAAM,KAAK,MAAM,KAAK,IAAI,KAAK,IAAI,QAAQ,CAAC,CAAC,CAAC,CAAC;CAS1E,OAAO;EACL,SAToC,IAAI,KAAK,IAAI,OAAO;GACxD,aAAa;GACb,UAAU,MAAM;GAChB,aAAa,KAAK,IAAI,KAAK,IAAI,QAAQ,MAAM,EAAG,CAAC,IAAI;GACrD,GAAG,WAAW;GACd,MAAM,UAAU,KAAM;EACxB,EAGiB,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;EACvD,YAAY;EACZ,YAAY;EACZ,WAAW,QAAQ;CACrB;AACF;AAiBA,SAAgB,eACd,SACA,SACA,OAAmB,CAAC,GACyB;CAC7C,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,IAAI,KAAK,WAAW;CAE1B,MAAM,KAAK,QAAQ,IAAI,QAAQ,MAAM,KAAK;CAC1C,MAAM,KAAK,QAAQ,IAAI,QAAQ,KAAK,KAAK;CAEzC,MAAM,YAAY,KAAK,IAAI,QAAQ,KAAK,MAAM;CAC9C,MAAM,SAAS,QAAQ,OAAO,KAAM;CACpC,MAAM,SAAS,QAAQ,OAAO,KAAM;CACpC,MAAM,IAAI,QAAQ,UAAU;CAE5B,MAAM,cAAc,IAAI,KAAK,SAAS;CACtC,MAAM,aAAa,IAAI,KAAK,UAAU,IAAI;CAE1C,QAAQ,IAAI,QAAQ,QAAQ,KAAK,WAAW;CAC5C,QAAQ,IAAI,QAAQ,OAAO,KAAK,UAAU;CAE1C,OAAO;EAAE;EAAa;CAAW;AACnC;AAsBA,SAAgB,0BACd,OACmB;CACnB,MAAM,aAAa,MAAM,cAAc;CACvC,MAAM,wBAAQ,IAAI,IAA2D;CAC7E,KAAK,MAAM,KAAK,MAAM,MAAM;EAC1B,MAAM,MAAM,MAAM,IAAI,EAAE,QAAQ,KAAK,CAAC;EACtC,IAAI,KAAK;GAAE,aAAa,EAAE;GAAa,OAAO,EAAE;EAAM,CAAC;EACvD,MAAM,IAAI,EAAE,UAAU,GAAG;CAC3B;CACA,MAAM,WAA8B,CAAC;CACrC,KAAK,MAAM,OAAO,MAAM,OAAO,GAC7B,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAC9B,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACvC,MAAM,IAAI,IAAI;EACd,MAAM,IAAI,IAAI;EACd,IAAI,EAAE,gBAAgB,EAAE,aAAa;EACrC,MAAM,SAAS,KAAK,IAAI,EAAE,QAAQ,EAAE,KAAK;EACzC,IAAI,UAAU,YACZ,SAAS,KAAK;GAAE,QAAQ,EAAE;GAAa,OAAO,EAAE;GAAa,MAAM;GAAM,QAAQ;EAAE,CAAC;OAC/E;GACL,MAAM,CAAC,QAAQ,SAAS,EAAE,QAAQ,EAAE,QAAQ,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC;GAC1D,SAAS,KAAK;IAAE,QAAQ,OAAO;IAAa,OAAO,MAAM;IAAa,QAAQ;GAAO,CAAC;EACxF;CACF;CAGJ,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC1OA,MAAa,0BAA0B;AA4LvC,MAAM,gCAAgC;AAiBtC,SAAS,KAAK,QAAgB,SAAwB;CACpD,MAAM,IAAI,MAAM,sBAAsB,OAAO,IAAI,SAAS;AAC5D;AAEA,SAAS,iBAAiB,QAAgB,WAAgD;CACxF,MAAM,MAAM,UAAU;CACtB,IAAI,QAAQ,MAAM;EAChB,IAAI,UAAU,UACZ,KACE,QACA,2FACF;EAEF,OAAO;GACL,SAAS;GACT,SAAS;GACT,UAAU;GACV,UAAU;GACV,iBAAiB;EACnB;CACF;CACA,MAAM,OAAO;EACX,SAAS,IAAI;EACb,UAAU,IAAI;EACd,UAAU,IAAI;EACd,iBAAiB,IAAI;CACvB;CACA,IAAI,IAAI,YAAY,MAAM;EACxB,IAAI,IAAI,oBAAoB,MAC1B,KACE,QACA,oFACF;EAEF,OAAO;GAAE,SAAS,IAAI,kBAAkB,YAAY;GAAe,GAAG;EAAK;CAC7E;CACA,IAAI,IAAI,aAAa,MAAM,OAAO;EAAE,SAAS;EAAqB,GAAG;CAAK;CAC1E,IAAI,CAAC,IAAI,aAAa,IAAI,YAAY,OAAO;EAAE,SAAS;EAAiB,GAAG;CAAK;CACjF,KAAK,QAAQ,oEAAoE;AACnF;AAEA,SAAS,oBACP,aACA,UACmF;CACnF,IAAI,gBAAgB,QAAQ,gBAAgB,KAAA,GAC1C,OAAO;EAAE,aAAa;EAAM,sBAAsB;EAAO,kBAAkB;CAAE;CAE/E,IAAI,YAAY,UAAU,UACxB,OAAO;EAAE;EAAa,sBAAsB;EAAO,kBAAkB,YAAY;CAAO;CAE1F,OAAO;EACL,aAAa,YAAY,MAAM,GAAG,QAAQ;EAC1C,sBAAsB;EACtB,kBAAkB,YAAY;CAChC;AACF;;;;;AAMA,SAAgB,wBAAwB,MAAuD;CAC7F,MAAM,EAAE,WAAW,OAAO,OAAO,WAAW;CAC5C,MAAM,SAAS,GAAG,KAAK,MAAM,GAAG,UAAU,OAAO,GAAG,UAAU;CAC9D,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,IAAI,UAAU,WAAW,MACvB,KACE,QACA,mBAAmB,UAAU,OAAO,YAAY,UAAU,SAAS,OAAO,uBAC5E;CAEF,IAAI,MAAM,YAAY,UAAU,QAC9B,KAAK,QAAQ,kBAAkB,MAAM,QAAQ,0BAA0B;CAEzE,IAAI,MAAM,eAAe,UAAU,WACjC,KAAK,QAAQ,oBAAoB,MAAM,WAAW,qBAAqB,UAAU,WAAW;CAE9F,IAAI,MAAM,WAAW,UAAU,WAC7B,KAAK,QAAQ,kBAAkB,MAAM,OAAO,uBAAuB,UAAU,WAAW;CAE1F,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;EACrC,MAAM,OAAO,MAAM;EACnB,IAAI,KAAK,YAAY,IAAI,GACvB,KAAK,QAAQ,4CAA4C,EAAE,eAAe,KAAK,SAAS;CAE5F;CACA,MAAM,IAAI,UAAU;CACpB,IAAI,IAAI,KAAK,IAAI,UAAU,WACzB,KAAK,QAAQ,eAAe,EAAE,iBAAiB,UAAU,WAAW;CAEtE,IAAI,CAAC,UAAU,mBAAmB,SAAS,CAAC,GAC1C,KACE,QACA,eAAe,EAAE,iCAAiC,UAAU,mBAAmB,KAAK,IAAI,EAAE,EAC5F;CAEF,MAAM,sBAAsB,CAC1B,GAAG,IAAI,IAAI,MAAM,iBAAiB,SAAS,MAAM,EAAE,kBAAkB,CAAC,CACxE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CACtB,KAAK,MAAM,YAAY,UAAU,oBAC/B,IAAI,CAAC,oBAAoB,SAAS,QAAQ,GACxC,KACE,QACA,kBAAkB,SAAS,oEAC7B;CAGJ,IAAI,WAAW,KAAA,GAAW;EACxB,IAAI,OAAO,MAAM,GAAG,KAAK,QAAQ,sBAAsB,OAAO,EAAE,eAAe,GAAG;EAClF,IAAI,OAAO,mBAAmB,UAAU,gBACtC,KACE,QACA,mCAAmC,OAAO,eAAe,aAAa,UAAU,gBAClF;EAEF,IAAI,OAAO,uBAAuB,UAAU,uBAC1C,KACE,QACA,uCAAuC,OAAO,mBAAmB,aAAa,UAAU,uBAC1F;EAEF,IAAI,OAAO,kBAAkB,WAAW,UAAU,mBAChD,KACE,QACA,0BAA0B,OAAO,kBAAkB,OAAO,gCAAgC,UAAU,mBACtG;EAEF,MAAM,UAAU,OAAO,kBAAkB,QAAQ,MAAM,EAAE,SAAS,qBAAqB,CAAC,CAAC;EACzF,IAAI,YAAY,UAAU,2BACxB,KACE,QACA,0BAA0B,QAAQ,uCAAuC,UAAU,2BACrF;CAEJ;CAEA,MAAM,UAAU,MAAM,IAAI;CAC1B,MAAM,kBAAoC,MAAM,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,UAAU;EACzE,QAAQ,KAAK;EACb,QAAQ,KAAK;EACb,GAAG,oBAAoB,KAAK,aAAa,mBAAmB;CAC9D,EAAE;CAEF,OAAO;EACL,QAAQ;EACR;EACA,QAAQ,UAAU;EAClB,QAAQ,UAAU;EAClB,MAAM;GACJ,OAAO,MAAM,SAAS;GACtB,OAAO,MAAM,SAAS;GACtB,UAAU,MAAM,aAAa;GAC7B,YAAY,MAAM,cAAc;GAChC,QAAQ,MAAM;GACd,WAAW,UAAU;EACvB;EACA,MAAM;GACJ,OAAO;GACP,WAAW,QAAQ;GACnB,oBAAoB,CAAC,GAAG,UAAU,kBAAkB,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;GAC1E;GACA,uBAAuB,UAAU;EACnC;EACA,YAAY;GACV,QAAQ;IAAE,OAAO;IAAG,KAAK;GAAE;GAC3B,OAAO;EACT;EACA,cAAc;GACZ,YAAY,UAAU;GACtB,iBAAiB,UAAU;GAC3B,gBAAgB,QAAQ,kBAAkB;GAC1C,WAAW,UAAU;GACrB,gBAAgB,UAAU;GAC1B,mBAAmB,UAAU;GAC7B,qBAAqB,UAAU;GAC/B,4BAA4B,UAAU;GACtC,2BAA2B,UAAU;GACrC,wBAAwB,QAAQ,qBAAqB;GACrD,UAAU,UAAU;GACpB,qBAAqB,UAAU;GAC/B,aAAa,QAAQ,eAAe;GACpC,QAAQ,UAAU;EACpB;EACA,KAAK,iBAAiB,QAAQ,SAAS;EACvC,YAAY;GACV,OAAO,KAAK;GACZ,kBAAkB,KAAK;GACvB,mBAAmB,KAAK;GACxB,YAAY,KAAK;GACjB,cAAc,KAAK;GACnB,WAAW,KAAK;GAChB,aAAa,KAAK;GAClB,OAAO,UAAU;GACjB,cAAc,UAAU;GACxB,KAAK,UAAU;GACf,WAAW,UAAU;GACrB,eAAe,QAAQ,OAAO,YAAY;GAC1C,WAAW,QAAQ,OAAO,QAAQ;EACpC;CACF;AACF;AAEA,SAAgB,0BAA0B,MAAqD;CAC7F,MAAM,UAAmC;EACvC,MAAM,KAAK;EACX,YAAY;EACZ,iBAAiB;EACjB,KAAK;GAAE,SAAS;GAAG,eAAe;GAAG,qBAAqB;GAAG,iBAAiB;EAAE;EAChF,UAAU,CAAC;CACb;CACA,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,aAAa,YAAY,QAAQ;EACzC,IAAI,IAAI,aAAa,cAAc,IAAI,aAAa,iBAAiB,QAAQ;EAC7E,QAAQ,IAAI,IAAI,IAAI,QAAQ;EAC5B,IAAI,SAAS,QAAQ,SAAS,IAAI;EAClC,IAAI,WAAW,KAAA,GAAW;GACxB,SAAS;IAAE,MAAM;IAAG,YAAY;IAAG,YAAY;GAAE;GACjD,QAAQ,SAAS,IAAI,UAAU;EACjC;EACA,OAAO;EACP,IAAI,IAAI,aAAa,YAAY,OAAO;EACxC,IAAI,IAAI,IAAI,YAAY,WAAW,OAAO;CAC5C;CACA,OAAO;AACT;AAEA,SAAgB,wBAAwB,MAAoC;CAC1E,OAAO,KAAK,KAAK,QAAQ,KAAK,UAAU,GAAG,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACvF;AAiCA,SAAS,OAAO,QAAwB;CACtC,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,MAAM,CAAC,CAAC,OAAO,KAAK;AACzD;AAEA,SAAS,SAAS,MAAc,MAAkD;CAChF,IAAI;CACJ,IAAI;EACF,SAAS,aAAa,IAAI;CAC5B,SAAS,OAAO;EACd,MAAM,IAAI,MACR,kCAAkC,KAAK,MAAM,KAAK,IAAK,MAAgB,SACzE;CACF;CACA,IAAI;EACF,OAAO;GAAE,OAAO,KAAK,MAAM,OAAO,SAAS,MAAM,CAAC;GAAG,QAAQ,OAAO,MAAM;EAAE;CAC9E,SAAS,OAAO;EACd,MAAM,IAAI,MACR,sBAAsB,KAAK,MAAM,KAAK,sBAAuB,MAAgB,SAC/E;CACF;AACF;AAYA,MAAM,0CAA0B,IAAI,IAAI,CAAC,uBAAuB,qBAAqB,CAAC;AAEtF,SAAS,sBAAsB,QAAgB,WAA+C;CAC5F,MAAM,OAAO,KAAK,QAAQ,GAAG,UAAU,OAAO,IAAI,UAAU,UAAU,qBAAqB;CAC3F,MAAM,SAAS,SAAS,MAAM,wBAAwB,UAAU,QAAQ,CAAC,CAAC;CAC1E,MAAM,cAAc,OAAO,qBAAqB,CAAC;CAGjD,KAAK,MAAM,cAAc,aACvB,IAAI,CAAC,wBAAwB,IAAI,WAAW,IAAI,GAC9C,MAAM,IAAI,MACR,sBAAsB,KAAK,QAAQ,WAAW,KAAK,wBAC7C,WAAW,KAAK,qBAAqB,CAAC,GAAG,uBAAuB,CAAC,CAAC,KAAK,IAAI,GACnF;CAGJ,OAAO;EACL,GAAG,OAAO;EACV,gBAAgB,OAAO;EACvB,oBAAoB,OAAO;EAC3B,gBAAgB,OAAO,kBAAkB;EACzC,mBAAmB;EACnB,aAAa,OAAO,MAAM,WAAW;EACrC,QAAQ;GACN,UAAU,OAAO,QAAQ,YAAY;GACrC,MAAM,OAAO,QAAQ,QAAQ;EAC/B;CACF;AACF;;;;;;AAOA,SAAgB,4BACd,QACyB;CACzB,MAAM,SAAS,SAAS,OAAO,iBAAiB,cAAc;CAC9D,MAAM,eAAe,OAAO;CAC5B,IAAI,CAAC,MAAM,QAAQ,aAAa,KAAK,KAAK,aAAa,MAAM,WAAW,GACtE,MAAM,IAAI,MAAM,sCAAsC,OAAO,gBAAgB,cAAc;CAE7F,IAAI,OAAO,aAAa,gBAAgB,YAAY,aAAa,YAAY,WAAW,GACtF,MAAM,IAAI,MACR,sCAAsC,OAAO,gBAAgB,oBAC/D;CAGF,MAAM,6BAAa,IAAI,IAAuE;CAC9F,MAAM,oBAAsE,CAAC;CAE7E,MAAM,iBAAiB,WAAmB;EACxC,MAAM,SAAS,OAAO,QAAQ;EAC9B,IAAI,WAAW,KAAA,GACb,MAAM,IAAI,MACR,sDAAsD,OAAO,kDAC/D;EAEF,IAAI,SAAS,WAAW,IAAI,MAAM;EAClC,IAAI,WAAW,KAAA,GAAW;GACxB,MAAM,SAAS,SAAS,OAAO,YAAY,sBAAsB,OAAO,EAAE;GAC1E,MAAM,UAAU,OAAO;GACvB,IAAI,CAAC,MAAM,QAAQ,OAAO,GACxB,MAAM,IAAI,MACR,yCAAyC,OAAO,OAAO,OAAO,WAAW,kBAC3E;GAEF,MAAM,2BAAW,IAAI,IAA4B;GACjD,KAAK,MAAM,SAAS,SAAS;IAC3B,IAAI,SAAS,IAAI,MAAM,OAAO,GAC5B,MAAM,IAAI,MACR,yCAAyC,OAAO,+BAA+B,MAAM,QAAQ,EAC/F;IAEF,SAAS,IAAI,MAAM,SAAS,KAAK;GACnC;GACA,SAAS;IAAE,QAAQ,OAAO;IAAQ;GAAS;GAC3C,WAAW,IAAI,QAAQ,MAAM;GAC7B,kBAAkB,UAAU;IAC1B,YAAY,OAAO;IACnB,cAAc,OAAO;IACrB,aAAa,OAAO;GACtB;EACF;EACA,OAAO;GAAE;GAAQ,GAAG;EAAO;CAC7B;CAEA,MAAM,OAA6B,CAAC;CACpC,KAAK,MAAM,aAAa,aAAa,OAAO;EAC1C,MAAM,EAAE,QAAQ,QAAQ,cAAc,aAAa,cAAc,UAAU,MAAM;EACjF,MAAM,QAAQ,SAAS,IAAI,UAAU,MAAM;EAC3C,IAAI,UAAU,KAAA,GACZ,MAAM,IAAI,MACR,sBAAsB,OAAO,MAAM,GAAG,UAAU,OAAO,GAAG,UAAU,OAAO,sBAAsB,OAAO,YAC1G;EAEF,MAAM,YAAY,KAAK,OAAO,aAAa,cAAc,UAAU,QAAQ,YAAY;EACvF,MAAM,YAAY,SAAS,WAAW,wBAAwB,UAAU,QAAQ;EAChF,MAAM,QAAQ,UAAU;EACxB,IAAI,CAAC,MAAM,QAAQ,KAAK,GACtB,MAAM,IAAI,MAAM,0CAA0C,UAAU,kBAAkB;EAExF,MAAM,SACJ,OAAO,WAAW,KAAA,IAAY,KAAA,IAAY,sBAAsB,OAAO,QAAQ,SAAS;EAC1F,KAAK,KACH,wBAAwB;GACtB;GACA;GACA;GACA,OAAO,OAAO;GACd,kBAAkB,aAAa;GAC/B,mBAAmB,OAAO;GAC1B,YAAY,OAAO;GACnB;GACA;GACA,aAAa,UAAU;GACvB;GACA,qBAAqB,OAAO;EAC9B,CAAC,CACH;CACF;CACA,KAAK,MAAM,GAAG,MAAM,EAAE,OAAO,cAAc,EAAE,MAAM,CAAC;CAEpD,OAAO;EACL;EACA,SAAS,0BAA0B,IAAI;EACvC,YAAY;GACV,OAAO,OAAO;GACd,iBAAiB,OAAO;GACxB,mBAAmB,OAAO;GAC1B,kBAAkB,aAAa;GAC/B,SAAS;EACX;CACF;AACF"}
|
|
1
|
+
{"version":3,"file":"rl.js","names":[],"sources":["../src/rl/adaptation-eval.ts","../src/rl/compute-curves.ts","../src/rl/contamination.ts","../src/rl/rollout-input.ts","../src/rl/exporters.ts","../src/rl/dataset.ts","../src/rl/corpus.ts","../src/rl/off-policy.ts","../src/rl/predictive-validity-researcher.ts","../src/rl/preferences.ts","../src/rl/process-reward.ts","../src/rl/rl-campaign.ts","../src/rl/run-record-adapters.ts","../src/rl/sim-fidelity.ts","../src/rl/tournament.ts","../src/rl/verified-findings-dataset.ts"],"sourcesContent":["/**\n * Sample-efficient adaptation evaluation.\n *\n * For foundation-model-based agents, the load-bearing capability isn't\n * raw end-state performance — it's *how fast the agent reaches that\n * performance from cold start*. The same model with a worse prompt that\n * adapts in 5 demonstrations beats the same model with a better prompt\n * that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)\n * reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …\n * in-context examples or fine-tune steps.\n *\n * This module ships:\n *\n * 1. `runAdaptationCurve` — given a runner that takes k demonstrations\n * and returns a score, produce the (k, score) curve.\n * 2. `compareAdaptationCurves` — paired comparison across two policies.\n * Returns per-k delta with bootstrap CIs and an \"area-under-curve\"\n * summary statistic.\n * 3. `firstPassK` — for pass/fail evaluation, the minimum k at which\n * the policy reliably passes (≥ pass-rate threshold over reps).\n *\n * Use cases:\n * - Compare two prompt designs that have similar end-state performance\n * but different in-context efficiency.\n * - Decide between fine-tuning and prompting based on adaptation cost.\n * - Detect when a policy \"memorizes\" k=0 inputs vs. genuinely adapts.\n */\n\nimport { makeRng } from '../statistics/internal'\n\nexport interface AdaptationRunner<S> {\n /**\n * Runs the policy on `scenario` with `k` demonstrations. Returns a\n * scalar score in [0, 1]. The runner is responsible for any caching;\n * the harness calls it once per (scenario, k, rep) cell.\n */\n run(args: { scenario: S; k: number; rep: number }): Promise<number>\n}\n\nexport interface RunAdaptationCurveOptions<S> {\n scenarios: S[]\n /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */\n ks?: number[]\n /** Reps per (scenario, k) cell. Default 3. */\n reps?: number\n runner: AdaptationRunner<S>\n /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */\n passThreshold?: number\n}\n\nexport interface AdaptationPoint {\n k: number\n meanScore: number\n passRate: number\n std: number\n n: number\n /** Per-scenario means at this k. */\n perScenario: Array<{ scenarioId: string; meanScore: number; passes: number; total: number }>\n}\n\nexport interface AdaptationCurve {\n points: AdaptationPoint[]\n /**\n * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`\n * tested reaches it.\n */\n firstPassK: number | null\n /**\n * Area under the (k, meanScore) curve, normalized by max-k. A\n * single-number summary of \"how well does this policy adapt from\n * cold-start to fully-conditioned.\" Higher = better adapter.\n */\n adaptationArea: number\n}\n\nexport async function runAdaptationCurve<S extends { scenarioId?: string }>(\n opts: RunAdaptationCurveOptions<S>,\n): Promise<AdaptationCurve> {\n const ks = opts.ks ?? [0, 1, 2, 4, 8, 16]\n const reps = opts.reps ?? 3\n const passThreshold = opts.passThreshold ?? 0.5\n const sortedKs = [...ks].sort((a, b) => a - b)\n\n const points: AdaptationPoint[] = []\n for (const k of sortedKs) {\n const perScenario: AdaptationPoint['perScenario'] = []\n const allScores: number[] = []\n let totalPasses = 0\n let totalAttempts = 0\n for (const scenario of opts.scenarios) {\n const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`\n const scores: number[] = []\n let passes = 0\n for (let r = 0; r < reps; r++) {\n const score = await opts.runner.run({ scenario, k, rep: r })\n scores.push(score)\n if (score >= passThreshold) passes++\n allScores.push(score)\n if (score >= passThreshold) totalPasses++\n totalAttempts++\n }\n const meanS = scores.reduce((s, v) => s + v, 0) / scores.length\n perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length })\n }\n const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length)\n const variance =\n allScores.length < 2\n ? 0\n : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1)\n points.push({\n k,\n meanScore,\n passRate: totalPasses / Math.max(1, totalAttempts),\n std: Math.sqrt(variance),\n n: allScores.length,\n perScenario,\n })\n }\n\n const firstPassK = points.find((p) => p.passRate >= passThreshold)?.k ?? null\n const maxK = sortedKs[sortedKs.length - 1] ?? 1\n // Trapezoidal area under the (k, meanScore) curve, normalized by k-range.\n let area = 0\n for (let i = 1; i < points.length; i++) {\n const x1 = points[i - 1]!.k\n const x2 = points[i]!.k\n const y1 = points[i - 1]!.meanScore\n const y2 = points[i]!.meanScore\n area += ((y1 + y2) / 2) * (x2 - x1)\n }\n const adaptationArea = maxK === 0 ? 0 : area / maxK\n\n return { points, firstPassK, adaptationArea }\n}\n\nexport interface CompareCurvesResult {\n perK: Array<{\n k: number\n deltaMean: number\n aLow: number\n aHigh: number\n bLow: number\n bHigh: number\n }>\n areaDelta: number\n firstPassKDelta: number | null\n /** Verdict: 'a_better' | 'b_better' | 'similar'. */\n verdict: 'a_better' | 'b_better' | 'similar'\n /** Rationale, ready to render. */\n rationale: string\n}\n\n/**\n * Paired comparison of two adaptation curves. Per-k deltas with 95%\n * bootstrap CIs (constructed from each curve's `perScenario` per-k means\n * — the bootstrap unit is the scenario, not the rep).\n */\nexport function compareAdaptationCurves(\n a: AdaptationCurve,\n b: AdaptationCurve,\n opts: { confidence?: number; bootstrapResamples?: number; seed?: number } = {},\n): CompareCurvesResult {\n const conf = opts.confidence ?? 0.95\n const resamples = opts.bootstrapResamples ?? 500\n const rng = makeRng(\n opts.seed,\n a.points.flatMap((point) => point.perScenario.map((cell) => cell.meanScore)),\n b.points.flatMap((point) => point.perScenario.map((cell) => cell.meanScore)),\n )\n\n const perK: CompareCurvesResult['perK'] = []\n for (const ap of a.points) {\n const bp = b.points.find((p) => p.k === ap.k)\n if (!bp) continue\n const aMeans = ap.perScenario.map((s) => s.meanScore)\n const bMeans = bp.perScenario.map((s) => s.meanScore)\n const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng)\n const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng)\n perK.push({\n k: ap.k,\n deltaMean: ap.meanScore - bp.meanScore,\n aLow: aCi.low,\n aHigh: aCi.high,\n bLow: bCi.low,\n bHigh: bCi.high,\n })\n }\n\n const areaDelta = a.adaptationArea - b.adaptationArea\n const firstPassKDelta =\n a.firstPassK !== null && b.firstPassK !== null\n ? b.firstPassK - a.firstPassK // smaller k for a means a adapts faster (positive delta)\n : null\n\n // Composite verdict: positive area delta + most per-k deltas in same\n // direction → that side wins. Within ε of zero on both → similar.\n const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length)\n let verdict: CompareCurvesResult['verdict']\n if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = 'similar'\n else if (meanDelta > 0 && areaDelta > 0) verdict = 'a_better'\n else if (meanDelta < 0 && areaDelta < 0) verdict = 'b_better'\n else verdict = 'similar'\n\n const rationale =\n `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` +\n (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : '')\n\n return { perK, areaDelta, firstPassKDelta, verdict, rationale }\n}\n\n/** First k at which the curve's per-scenario pass rate reliably hits the threshold. */\nexport function firstPassK(curve: AdaptationCurve, threshold = 0.5): number | null {\n return curve.points.find((p) => p.passRate >= threshold)?.k ?? null\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction bootstrapMeanCi(\n xs: number[],\n resamples: number,\n confidence: number,\n rng: () => number,\n): { low: number; high: number } {\n if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 }\n const samples = new Array<number>(resamples)\n for (let b = 0; b < resamples; b++) {\n let sum = 0\n for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)]!\n samples[b] = sum / xs.length\n }\n samples.sort((a, b) => a - b)\n const alpha = 1 - confidence\n return {\n low: samples[Math.floor((alpha / 2) * resamples)]!,\n high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]!,\n }\n}\n","/**\n * Test-time compute scaling curves.\n *\n * The test-time-compute frontier paper (Snell et al. 2024) and the\n * subsequent o1-style scaling work both show that LLM-agent capability\n * is a function of the compute budget at inference, not just of the\n * training run. The right way to characterize a candidate is therefore\n * a *curve* — score at compute budgets {1×, 4×, 16×, …} — not a single\n * point.\n *\n * This module ships:\n *\n * 1. The compute-curve harness — `runComputeCurve(runner, budgets)` —\n * that evaluates one candidate at a sequence of compute budgets\n * and returns the (compute, score) curve.\n * 2. A best-of-N evaluator — `bestOfN(runner, n, scoreFn)` — the\n * simplest test-time-compute scaling primitive: sample N\n * independent rollouts, return the best.\n * 3. A self-consistency evaluator — `selfConsistency(runner, n)` —\n * the majority-vote variant of best-of-N for tasks with a small\n * categorical answer space.\n * 4. Pareto-frontier extraction over multiple candidates — given\n * (candidate, compute, score) tuples, return the set of\n * candidate-compute combinations that aren't dominated.\n *\n * Caveat: \"compute\" here is the caller's notion of a compute unit. For\n * agent eval that's typically wall-time × parallelism, or token budget,\n * or LLM-call count. We accept whatever the caller provides; the curve\n * is on whatever axis they pick.\n */\n\nimport { ValidationError } from '../errors'\n\nexport interface ComputeCurveBudget {\n /** Identifier — for the report. Common: '1x', '4x', '16x'. */\n id: string\n /** Numeric value on the chosen axis (tokens, calls, USD, ms — caller picks). */\n cost: number\n /** Free-form metadata (the caller can carry per-budget config). */\n meta?: Record<string, unknown>\n}\n\nexport interface ComputeCurvePoint {\n budgetId: string\n cost: number\n score: number\n /** Number of underlying samples used at this budget. */\n samples: number\n /** Optional spread / variance information. */\n std?: number\n /** Any extra metrics the runner returned. */\n metrics?: Record<string, number>\n}\n\nexport interface ComputeCurve {\n candidateId: string\n points: ComputeCurvePoint[]\n /** Rough exponent fit: score ≈ a + b * log(cost). Useful for \"how steep is the curve?\" */\n logSlope: number | null\n /** Best (highest-score) point on the curve. */\n best: ComputeCurvePoint\n}\n\nexport interface RunComputeCurveOptions {\n candidateId: string\n budgets: ComputeCurveBudget[]\n /**\n * Run the candidate at one budget. Returns the realized score plus\n * optional spread + extra metrics.\n */\n runAtBudget: (budget: ComputeCurveBudget) => Promise<{\n score: number\n samples: number\n std?: number\n metrics?: Record<string, number>\n }>\n}\n\nexport async function runComputeCurve(opts: RunComputeCurveOptions): Promise<ComputeCurve> {\n const points: ComputeCurvePoint[] = []\n for (const budget of opts.budgets) {\n const r = await opts.runAtBudget(budget)\n points.push({\n budgetId: budget.id,\n cost: budget.cost,\n score: r.score,\n samples: r.samples,\n std: r.std,\n metrics: r.metrics,\n })\n }\n const sorted = [...points].sort((a, b) => a.cost - b.cost)\n const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null\n const best = points.reduce((a, b) => (b.score > a.score ? b : a))\n return { candidateId: opts.candidateId, points: sorted, logSlope, best }\n}\n\nexport interface ComputeBestOfNOptions<O> {\n /** Number of independent samples to draw. */\n n: number\n /** Sampler — produces one rollout. */\n sample: (sampleIdx: number) => Promise<O>\n /** Score one rollout. */\n scoreFn: (rollout: O) => Promise<number> | number\n}\n\nexport interface ComputeBestOfNResult<O> {\n best: O\n bestScore: number\n scores: number[]\n meanScore: number\n /** Index of the best rollout, for diagnostics. */\n bestIndex: number\n}\n\n/** The simplest test-time scaling primitive. */\nexport async function bestOfN<O>(opts: ComputeBestOfNOptions<O>): Promise<ComputeBestOfNResult<O>> {\n if (opts.n <= 0) throw new ValidationError('bestOfN: n must be > 0')\n const rollouts: O[] = []\n const scores: number[] = []\n for (let i = 0; i < opts.n; i++) {\n const r = await opts.sample(i)\n rollouts.push(r)\n scores.push(await opts.scoreFn(r))\n }\n let bestIndex = 0\n for (let i = 1; i < scores.length; i++) if (scores[i]! > scores[bestIndex]!) bestIndex = i\n const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length\n return {\n best: rollouts[bestIndex]!,\n bestScore: scores[bestIndex]!,\n scores,\n meanScore,\n bestIndex,\n }\n}\n\nexport interface SelfConsistencyOptions<O> {\n n: number\n sample: (sampleIdx: number) => Promise<O>\n /** Extract the canonical answer key (string) from a rollout. */\n answerKey: (rollout: O) => string\n}\n\nexport interface SelfConsistencyResult<O> {\n /** Modal answer (the majority vote). */\n answer: string\n /** Fraction of samples voting for the modal answer in [0, 1]. */\n agreement: number\n /** Histogram of all answers. */\n histogram: Record<string, number>\n /** A representative rollout that voted for the modal answer. */\n representative: O\n /** All rollouts. */\n rollouts: O[]\n}\n\n/**\n * Self-consistency / majority-vote test-time scaling. For tasks with a\n * small categorical answer space (math problems, multiple choice).\n */\nexport async function selfConsistency<O>(\n opts: SelfConsistencyOptions<O>,\n): Promise<SelfConsistencyResult<O>> {\n if (opts.n <= 0) throw new ValidationError('selfConsistency: n must be > 0')\n const rollouts: O[] = []\n const histogram: Record<string, number> = {}\n for (let i = 0; i < opts.n; i++) {\n const r = await opts.sample(i)\n rollouts.push(r)\n const key = opts.answerKey(r)\n histogram[key] = (histogram[key] ?? 0) + 1\n }\n let answer = ''\n let max = -1\n for (const [k, v] of Object.entries(histogram)) {\n if (v > max) {\n max = v\n answer = k\n }\n }\n const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0]!\n return {\n answer,\n agreement: max / opts.n,\n histogram,\n representative,\n rollouts,\n }\n}\n\n/**\n * Pareto frontier over (candidate, compute, score) tuples. A point is on\n * the frontier iff no other point dominates it in both score (higher\n * better) and cost (lower better). Returns the frontier sorted ascending\n * by cost.\n */\nexport interface ParetoPointInput {\n candidateId: string\n budgetId: string\n cost: number\n score: number\n}\n\nexport function paretoFrontier(points: ParetoPointInput[]): ParetoPointInput[] {\n const onFrontier: ParetoPointInput[] = []\n for (const p of points) {\n const dominated = points.some(\n (q) =>\n q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score),\n )\n if (!dominated) onFrontier.push(p)\n }\n return onFrontier.sort((a, b) => a.cost - b.cost)\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction fitLogSlope(points: ComputeCurvePoint[]): number {\n // OLS slope of score on log(cost). Used as a single-number summary of\n // how much marginal compute helps. Positive = score improves with\n // compute; near-zero = capability ceiling reached.\n const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)))\n const ys = points.map((p) => p.score)\n const n = xs.length\n const mx = xs.reduce((s, x) => s + x, 0) / n\n const my = ys.reduce((s, y) => s + y, 0) / n\n let num = 0\n let den = 0\n for (let i = 0; i < n; i++) {\n num += (xs[i]! - mx) * (ys[i]! - my)\n den += (xs[i]! - mx) ** 2\n }\n return den === 0 ? 0 : num / den\n}\n","/**\n * Contamination probe — held-out perturbation tests.\n *\n * The bug class: once a benchmark scenario set is published, models train\n * on it, and your scores become invalid. SWE-Bench-Verified, GPQA, and\n * MMLU-Pro all exist because their predecessors got contaminated within\n * months. The right defense is to keep a held-out *perturbed* version of\n * every scenario — same task, slightly different surface — and check\n * whether scores diverge significantly. Genuine capability transfers; rote\n * memorization doesn't.\n *\n * This module ships the probe contract:\n *\n * 1. A `ScenarioPerturbation` strategy type — function that produces a\n * perturbed scenario from an original.\n * 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs\n * both halves and reports per-scenario score divergence + a global\n * contamination verdict via paired Wilcoxon.\n * 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,\n * `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the\n * task's structural difficulty while breaking surface memorization.\n *\n * The verdict is conservative: if the perturbed-vs-original score\n * difference is statistically significant (BH-adjusted p < 0.05) AND\n * the median drop is > 5 percentage points, we flag *contamination\n * suspected*. False positives are possible (the perturbation might\n * actually be harder); the default is to flag for review, not to\n * autoreject.\n */\n\nimport { ValidationError } from '../errors'\nimport { benjaminiHochberg, wilcoxonSignedRank } from '../statistics'\nimport { mulberry32 } from '../statistics/random'\n\nexport type ScenarioPerturbationKind =\n | 'rename_variables'\n | 'shuffle_order'\n | 'paraphrase'\n | 'inject_irrelevant_clause'\n | 'custom'\n\nexport interface ScenarioPerturbation<S> {\n kind: ScenarioPerturbationKind\n /** Apply to one scenario, return its perturbed sibling. */\n apply: (scenario: S) => Promise<S> | S\n /** Optional id — for the report. */\n id?: string\n}\n\nexport interface ContaminationProbeInput<S> {\n /** Identity of every scenario. The probe's `runFingerprint` keys on these. */\n scenarioId: (s: S) => string\n /** Original scenarios. */\n originals: S[]\n /**\n * Either pre-computed perturbations (one per original, same order) OR a\n * `perturbation` strategy that synthesizes them on the fly.\n */\n perturbed?: S[]\n perturbation?: ScenarioPerturbation<S>\n /**\n * Run the policy/agent against one scenario and return a scalar score\n * in [0, 1]. The probe doesn't care what the policy is — that's the\n * caller's contract.\n */\n scoreFn: (s: S) => Promise<number>\n}\n\nexport interface ContaminationProbeOptions {\n /** Drop scores below this from the probe; treats partial failures separately. Default 0. */\n scoreFloor?: number\n /**\n * BH-FDR threshold for declaring contamination on each per-scenario\n * delta. Default 0.05.\n */\n fdr?: number\n /**\n * Minimum median per-scenario drop to flag global contamination. Default\n * 0.05 (5 percentage points). Smaller drops may be noise.\n */\n minMedianDrop?: number\n}\n\nexport interface ContaminationProbeReport {\n perScenario: Array<{\n scenarioId: string\n originalScore: number\n perturbedScore: number\n delta: number // perturbed - original (negative = drop)\n /** Per-scenario q-value (single-test BH for a single scenario). Mainly for display. */\n qValue: number\n }>\n /** Wilcoxon paired-test on the deltas. */\n pairedTest: { w: number; p: number }\n medianDelta: number\n meanDelta: number\n contaminationSuspected: boolean\n reason: string\n /** Number of scenarios processed. */\n n: number\n}\n\nexport async function runContaminationProbe<S>(\n input: ContaminationProbeInput<S>,\n opts: ContaminationProbeOptions = {},\n): Promise<ContaminationProbeReport> {\n const fdr = opts.fdr ?? 0.05\n const minMedianDrop = opts.minMedianDrop ?? 0.05\n const floor = opts.scoreFloor ?? 0\n\n if (!input.perturbed && !input.perturbation) {\n throw new ValidationError(\n 'runContaminationProbe: must supply either `perturbed` or `perturbation`.',\n )\n }\n const perturbed: S[] =\n input.perturbed ?? (await Promise.all(input.originals.map((s) => input.perturbation!.apply(s))))\n if (perturbed.length !== input.originals.length) {\n throw new ValidationError(\n `runContaminationProbe: perturbed length ${perturbed.length} ≠ originals ${input.originals.length}`,\n )\n }\n\n // Score both halves.\n const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)))\n const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)))\n\n const perScenario = input.originals.map((s, i) => ({\n scenarioId: input.scenarioId(s),\n originalScore: origScores[i]!,\n perturbedScore: pertScores[i]!,\n delta: pertScores[i]! - origScores[i]!,\n qValue: NaN,\n }))\n\n // Drop scenarios below the floor (partial failures we don't trust).\n const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor)\n if (valid.length < 4) {\n return {\n perScenario,\n pairedTest: { w: 0, p: 1 },\n medianDelta: 0,\n meanDelta: 0,\n contaminationSuspected: false,\n reason: `insufficient valid scenarios (n=${valid.length}, need ≥ 4)`,\n n: valid.length,\n }\n }\n\n const origValid = valid.map((p) => p.originalScore)\n const pertValid = valid.map((p) => p.perturbedScore)\n const pairedTest = wilcoxonSignedRank(origValid, pertValid)\n const deltas = valid.map((p) => p.delta)\n const sortedDeltas = [...deltas].sort((a, b) => a - b)\n const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)]!\n const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length\n\n // Per-scenario q-values via BH on a synthetic per-scenario p-value\n // (one-sample bootstrap; we use the absolute delta normalized by median\n // as a coarse signal — this is a display aid, the load-bearing test\n // is the global Wilcoxon).\n const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)))\n const { qValues } = benjaminiHochberg(pseudoP, fdr)\n for (let i = 0; i < valid.length; i++) {\n const v = valid[i]!\n const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId)\n if (idx >= 0) perScenario[idx]!.qValue = qValues[i]!\n }\n\n const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop\n const reason = contaminationSuspected\n ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} ≥ ${minMedianDrop}`\n : pairedTest.p >= fdr\n ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})`\n : `significant but small effect (median delta ${median.toFixed(4)})`\n\n return {\n perScenario,\n pairedTest,\n medianDelta: median,\n meanDelta: mean,\n contaminationSuspected,\n reason,\n n: valid.length,\n }\n}\n\n// ── Stock perturbations ──────────────────────────────────────────────────\n\n/**\n * Identifier-rename perturbation for code/text scenarios. Replaces every\n * occurrence of the listed identifiers with synthesized aliases. Use when\n * the scenario's structural difficulty is independent of variable names\n * (e.g. SWE-Bench-style coding tasks).\n */\nexport function renameVariables<S extends { prompt: string }>(\n identifiers: string[],\n rename: (name: string, idx: number) => string = (n, i) => `${n}_${((i % 26) + 10).toString(36)}`,\n): ScenarioPerturbation<S> {\n return {\n kind: 'rename_variables',\n apply(scenario) {\n let prompt = scenario.prompt\n identifiers.forEach((id, i) => {\n const replacement = rename(id, i)\n const re = new RegExp(`\\\\b${escapeRegex(id)}\\\\b`, 'g')\n prompt = prompt.replace(re, replacement)\n })\n return { ...scenario, prompt }\n },\n }\n}\n\n/**\n * Order-shuffle perturbation. Reshuffles a list-shaped section of the\n * prompt (for QA scenarios that present options A/B/C/D — answer depends\n * on the option labels, not order). Caller provides the section extractor.\n */\nexport function shuffleOrder<S extends { prompt: string }>(\n shuffleSection: (prompt: string, rng: () => number) => string,\n seed: number,\n): ScenarioPerturbation<S> {\n const rng = mulberry32(seed)\n return {\n kind: 'shuffle_order',\n apply(scenario) {\n const newPrompt = shuffleSection(scenario.prompt, rng)\n return { ...scenario, prompt: newPrompt }\n },\n }\n}\n\n/**\n * Inject-irrelevant-clause perturbation. Adds a benign sentence that\n * shouldn't change the answer. Tests for \"did the model just memorize\n * the input string.\"\n */\nexport function injectIrrelevantClause<S extends { prompt: string }>(\n clause: string,\n position: 'prefix' | 'suffix' = 'prefix',\n): ScenarioPerturbation<S> {\n return {\n kind: 'inject_irrelevant_clause',\n apply(scenario) {\n const prompt =\n position === 'prefix' ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`\n return { ...scenario, prompt }\n },\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction escapeRegex(s: string): string {\n return s.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n}\n","/**\n * Shared checks for trainer exports over canonical minted rollout lines.\n *\n * Exporters accept only `MintedRolloutLine[]`. Callers convert run records with\n * `mintRolloutRows` before deriving preferences or trainer files.\n */\n\nimport { assertRewardGate, type MintedRolloutLine } from '../rollout/schema'\n\n/**\n * The line's reward, with `null` meaning \"no verdict exists\" and never \"scored\n * zero\".\n *\n * The wire contract represents an absent verdict directly as `reward: null`.\n */\nexport function trainableLineReward(line: MintedRolloutLine): number | null {\n // The single reward reader on the `rl/` side, so the runtime invariant check\n // sits here and covers GRPO rows, SFT row metadata, preference ordering, and\n // the published datasheet's reward distribution in one place.\n assertRewardGate(line, 'trainable reward')\n const { reward } = line.outcome\n if (reward === null || !Number.isFinite(reward)) return null\n return reward\n}\n\n/** The anti-Goodhart flag as it travels on the line. */\nexport function isLineRealnessGated(line: MintedRolloutLine): boolean {\n return line.outcome.realness_gated === true\n}\n\n/**\n * The minted lines behind a LINE-LESS training artifact.\n *\n * `PreferenceTriple`, `PrmTrainingTriple` and `StepReward` all carry a bare\n * reward number plus run ids, and nothing that says whether those runs faked\n * their success. An exporter over them therefore has no way, from its input\n * alone, to learn that its chosen side is a run the gate flagged — it will\n * happily emit the gaming trajectory as the preferred one. Supplying the lines\n * is what gives it eyes.\n */\nexport interface RolloutLineContext {\n /**\n * Minted lines for every INVOCATION the artifacts reference.\n *\n * Not \"one line per run\": `tangle.rollout.v1` models many invocations per\n * `run_id` — that is what `rollout_id` and `parent_rollout_id` are for, and\n * `supervisorRunRolloutLines` emits a supervisor node plus one per worker, all\n * sharing a single `run_id`. A reference is resolved against `rollout_id`\n * first and falls back to `run_id` only when that run has exactly one\n * invocation; see `resolveInvocation`.\n */\n lines: MintedRolloutLine[]\n}\n\n/** How one exporter names itself and its context type in the failure messages. */\nexport interface LineContextRequirement {\n /** Exporter label, e.g. `'DPO export'`. */\n exporter: string\n /** The context type the caller must pass, e.g. `'DpoLineContext'`. */\n contextType: string\n /** Why this exporter cannot see the gate without lines. One sentence. */\n because: string\n}\n\n/**\n * How ONE reference on a line-less artifact resolved against the supplied lines.\n *\n * `'ambiguous'` is a real answer, not an error path: the id named more than one\n * invocation and there is no way to tell which one the artifact meant.\n */\ntype Resolution =\n | { kind: 'resolved'; line: MintedRolloutLine }\n | { kind: 'ambiguous'; count: number }\n | { kind: 'missing' }\n\n/**\n * Both identity keys a line-less artifact might be carrying, built once.\n *\n * `rollout_id` is the INVOCATION id and is the key that answers the question\n * these exporters ask (\"was the run behind this side of the pair gamed?\").\n * `run_id` is an EPISODE id: `mintRolloutRows` writes it equal to `rollout_id`\n * for a solo run, but `supervisorRunRolloutLines` shares one `run_id` across a\n * supervisor node and every worker it spawned. The published artifacts\n * (`PreferenceTriple.chosenRunId`, `StepReward.runId`,\n * `PrmTrainingTriple.prefixRunId`) name run ids, so the fallback has to exist —\n * but only where it cannot be wrong.\n */\ninterface InvocationIndex {\n byRollout: Map<string, MintedRolloutLine[]>\n byRun: Map<string, MintedRolloutLine[]>\n}\n\nfunction push(index: Map<string, MintedRolloutLine[]>, key: string, line: MintedRolloutLine): void {\n const existing = index.get(key)\n if (existing === undefined) index.set(key, [line])\n else existing.push(line)\n}\n\nfunction invocationIndex(lines: readonly MintedRolloutLine[]): InvocationIndex {\n const byRollout = new Map<string, MintedRolloutLine[]>()\n const byRun = new Map<string, MintedRolloutLine[]>()\n for (const line of lines) {\n push(byRollout, line.rollout_id, line)\n push(byRun, line.run_id, line)\n }\n return { byRollout, byRun }\n}\n\n/**\n * Resolve one referenced id to exactly ONE invocation, or refuse to guess.\n *\n * The previous implementation was `new Map(lines.map((l) => [l.run_id, l]))`,\n * which is LAST-WINS: with a gated supervisor node and an ungated worker sharing\n * a `run_id`, the answer depended on which one appeared later in the array, so\n * `[gatedRoot, worker, rival]` emitted the gamed trajectory as `chosen` and\n * simply reordering the same input suppressed it. An order-dependent security\n * property passes every test whose fixture happens to be ordered favourably,\n * which is the worst possible failure mode for a gate.\n *\n * The rule that removes order from the answer: an id resolves only when it names\n * one invocation. `rollout_id` is tried first because it IS the invocation id;\n * `run_id` is accepted only when the run holds a single invocation, and a\n * cross-index disagreement (an id that is one line's `rollout_id` and a\n * different line's `run_id`) is ambiguous rather than silently preferring\n * either.\n */\nfunction resolveInvocation(index: InvocationIndex, id: string): Resolution {\n const rollouts = index.byRollout.get(id) ?? []\n const runs = index.byRun.get(id) ?? []\n if (rollouts.length > 1) return { kind: 'ambiguous', count: rollouts.length }\n const exact = rollouts[0]\n if (exact !== undefined) {\n if (runs.some((line) => line !== exact)) {\n return { kind: 'ambiguous', count: 1 + runs.filter((line) => line !== exact).length }\n }\n return { kind: 'resolved', line: exact }\n }\n if (runs.length > 1) return { kind: 'ambiguous', count: runs.length }\n const only = runs[0]\n return only === undefined ? { kind: 'missing' } : { kind: 'resolved', line: only }\n}\n\n/** What the admission rule did, item by item — the count a caller has to be able to see. */\nexport interface AdmissionAudit<T> {\n admitted: T[]\n /** Items dropped because a referenced invocation was realness-gated. */\n gatedDrops: number\n /** Items dropped because a referenced id named more than one invocation. */\n ambiguousDrops: number\n /** Each ambiguous id and how many invocations it named, deduped, first-seen order. */\n ambiguous: Array<{ id: string; invocations: number }>\n}\n\n/**\n * THE admission rule for every exporter whose input is line-less — one\n * implementation, because two siblings over the same input class with different\n * gating is the defect being eliminated, and it has now happened twice\n * (`toPrmRows` hardened while `toDpoRows` was left open; `toGrpoRows`'\n * `rewardOf` gated while `extractPreferences`' identically-named hook was not).\n *\n * Fail-closed in five steps:\n * 1. No context at all → throw. A two-argument call used to be accepted and\n * produced rows with no gate applied whatsoever.\n * 2. A referenced id with NO line → throw. Its gate status is unknown, and\n * unknown is not clean. Thrown rather than dropped because it means the\n * caller did not supply the context it was asked for, which is a defect in\n * the call, not in the data.\n * 3. A referenced id naming MORE THAN ONE invocation → DROP the item and count\n * it. Dropped rather than thrown because, unlike (2), this is ordinary data\n * — a supervision episode legitimately holds a supervisor invocation and\n * several workers under one `run_id` — and throwing would make these\n * exporters unusable on any supervisor corpus, whose only workaround is for\n * the caller to hand-filter `context.lines` down to one line per run. That\n * workaround IS the leak, performed by hand. The count is surfaced by\n * `admitUngatedByInvocation` so the drop is never silent.\n * 4. Every resolved line goes through `assertRewardGate`, so the line-less\n * exporters compose the same check list as the waist exporters instead of\n * relying on `realness_gated` alone (which is one of three checks).\n * 5. Either side realness-gated → DROP the item. Dropped rather than zeroed\n * because these shapes have no honest zero: a preference pair is a\n * statement that one trajectory is better than another, and a gamed\n * trajectory belongs on neither side of it — as the chosen one it teaches\n * the gaming move outright, and as the rejected one it still ships the\n * gaming trajectory's text into the training file as a contrast example\n * nobody asked for.\n *\n * `inspect` is the per-exporter extra check (PRM's trajectory-completeness\n * rules). It runs on every resolved line before any item is admitted, so the\n * whole batch fails before a single row is built.\n *\n * Pure: it reports what it dropped and prints nothing.\n */\nexport function auditInvocationAdmission<T>(\n items: readonly T[],\n idsOf: (item: T) => readonly string[],\n context: RolloutLineContext | undefined | null,\n requirement: LineContextRequirement,\n inspect?: (line: MintedRolloutLine) => void,\n): AdmissionAudit<T> {\n if (context === undefined || context === null) {\n throw new Error(\n `${requirement.exporter}: a ${requirement.contextType} is required — ${requirement.because} Pass \\`{ lines: (await mintRolloutRows(...)).rows }\\`.`,\n )\n }\n const index = invocationIndex(context.lines)\n const audit: AdmissionAudit<T> = {\n admitted: [],\n gatedDrops: 0,\n ambiguousDrops: 0,\n ambiguous: [],\n }\n for (const item of items) {\n const lines: MintedRolloutLine[] = []\n let ambiguous = false\n for (const id of idsOf(item)) {\n const resolution = resolveInvocation(index, id)\n if (resolution.kind === 'missing') {\n throw new Error(\n `${requirement.exporter}: no rollout line supplied for run ${id} — its realness gate and capture quality are unknown`,\n )\n }\n if (resolution.kind === 'ambiguous') {\n ambiguous = true\n if (!audit.ambiguous.some((entry) => entry.id === id)) {\n audit.ambiguous.push({ id, invocations: resolution.count })\n }\n continue\n }\n lines.push(resolution.line)\n }\n for (const line of lines) {\n assertRewardGate(line, requirement.exporter)\n inspect?.(line)\n }\n if (ambiguous) {\n audit.ambiguousDrops++\n continue\n }\n if (lines.some(isLineRealnessGated)) {\n audit.gatedDrops++\n continue\n }\n audit.admitted.push(item)\n }\n return audit\n}\n\n/**\n * `auditInvocationAdmission` for the exporters, which return rows and have\n * nowhere to put a count.\n *\n * The ambiguous drops are announced rather than swallowed: a caller who asked\n * for N pairs and silently received N-k has no way to notice that a chunk of\n * their preference data quietly evaporated, and \"the training set got smaller\n * for a reason nobody printed\" is the same class of invisible failure as the\n * gate that never ran. A gated drop is NOT announced — that one is the gate\n * doing exactly its job, on the population the caller already knows is flagged.\n */\nexport function admitUngatedByInvocation<T>(\n items: readonly T[],\n idsOf: (item: T) => readonly string[],\n context: RolloutLineContext | undefined | null,\n requirement: LineContextRequirement,\n inspect?: (line: MintedRolloutLine) => void,\n): T[] {\n const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect)\n if (audit.ambiguousDrops > 0) {\n const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(', ')\n console.warn(\n `[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more ` +\n 'than one invocation in the supplied lines, so the realness gate cannot be read for the ' +\n 'invocation the artifact meant. Reference the `rollout_id` instead of the `run_id`, or ' +\n 'supply a context holding one invocation per run.',\n )\n }\n return audit.admitted\n}\n","/**\n * Trainer-format exporters.\n *\n * agent-eval produces canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`,\n * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume\n * different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI\n * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own\n * JSONL conventions. Rather than ship N adapters, this module ships the\n * canonical formats most production pipelines accept and ergonomic helpers\n * for the rest.\n *\n * Shapes:\n * - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed\n * by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.\n * - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.\n * Consumed by prime-rl GRPO, verl, OpenRLHF.\n * - **SFT** — `{messages[]}` JSONL with chosen completion as the final\n * assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,\n * Anthropic finetuning.\n * - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.\n * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.\n *\n * Why ship this in agent-eval rather than a separate adapter package: the\n * canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`, etc.) are\n * agent-eval's contract; without first-party exporters consumers reverse-\n * engineer the mapping every release. The exporters codify it.\n *\n * The exporters take callbacks for any field that isn't on the canonical\n * artifact (specifically: prompt + completion text, since the package\n * stores only their hashes by design — full text is the consumer's\n * trace store / raw event log).\n *\n * Every exporter that produces a training row accepts canonical minted rollout\n * lines. Convert run records once with `mintRolloutRows`; downstream transforms\n * then share one reward, split, and authenticity contract.\n */\n\nimport { isSplitEligible } from '../rollout/exporters'\nimport { assertRewardGate, type MintedRolloutLine, type RolloutSplit } from '../rollout/schema'\nimport type { PreferenceTriple } from './preferences'\nimport type { PrmTrainingTriple, StepReward } from './process-reward'\nimport {\n admitUngatedByInvocation,\n isLineRealnessGated,\n type LineContextRequirement,\n type RolloutLineContext,\n trainableLineReward,\n} from './rollout-input'\n\nexport type { RolloutLineContext } from './rollout-input'\n\n// ── DPO / IPO / KTO ──────────────────────────────────────────────────────\n\nexport interface DpoLookups {\n /** Resolve the prompt text for a run (typically from a trace store / raw event sink). */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the assistant completion text for a run. */\n completionOf: (runId: string) => string | Promise<string>\n}\n\nexport interface DpoExportRow {\n prompt: string\n chosen: string\n rejected: string\n /** Carried-through margin. Some KTO / IPO variants use this. */\n margin?: number\n /** Free-form metadata for downstream filtering / sharding. */\n meta?: Record<string, unknown>\n}\n\n/** The minted lines for the runs a `PreferenceTriple` names on each side. */\nexport type DpoLineContext = RolloutLineContext\n\nconst DPO_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'DPO export',\n contextType: 'DpoLineContext',\n because:\n 'a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair — which is DPO trained to PREFER the gaming trajectory.',\n}\n\n/**\n * Convert preference triples to TRL-compatible DPO rows. The shape\n * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset\n * entry; every major DPO trainer accepts it.\n *\n * `context` is REQUIRED, and for the same reason it is required on the sibling\n * `toPrmRows`: a triple is a line-less artifact. It names two run ids and a\n * margin, and nothing on it says whether either run was flagged as gamed —\n * so a two-argument call applied NO gate at all and emitted the row verbatim,\n * reachable straight through the published bundle builder\n * (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).\n * Triples whose chosen or rejected side is realness-gated are dropped; a triple\n * naming a run with no supplied line is refused. See `admitUngatedByInvocation` for\n * why dropping, not zeroing, is the right disposition for a preference pair.\n */\nexport async function toDpoRows(\n triples: PreferenceTriple[],\n lookups: DpoLookups,\n context: DpoLineContext,\n): Promise<DpoExportRow[]> {\n const admitted = admitUngatedByInvocation(\n triples,\n (t) => [t.chosenRunId, t.rejectedRunId],\n context,\n DPO_CONTEXT_REQUIREMENT,\n )\n const out: DpoExportRow[] = []\n for (const t of admitted) {\n const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([\n Promise.resolve(lookups.promptOf(t.chosenRunId)),\n Promise.resolve(lookups.promptOf(t.rejectedRunId)),\n Promise.resolve(lookups.completionOf(t.chosenRunId)),\n Promise.resolve(lookups.completionOf(t.rejectedRunId)),\n ])\n if (chosenPrompt !== rejectedPrompt) {\n throw new Error(\n `toDpoRows: preference \"${t.chosenRunId}\"/\"${t.rejectedRunId}\" resolves to different prompts`,\n )\n }\n out.push({\n prompt: chosenPrompt,\n chosen,\n rejected,\n margin: t.marginScore,\n meta: {\n scenarioId: t.scenarioId,\n chosenVariantId: t.chosenVariantId,\n rejectedVariantId: t.rejectedVariantId,\n chosenRunId: t.chosenRunId,\n rejectedRunId: t.rejectedRunId,\n chosenModel: t.meta.chosenModel,\n rejectedModel: t.meta.rejectedModel,\n },\n })\n }\n return out\n}\n\n/** Serialize DPO rows as JSONL. One line per row. */\nexport function toDpoJsonl(rows: DpoExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── GRPO offline ─────────────────────────────────────────────────────────\n\nexport interface TrainingLineSelectionOptions {\n /** Include held-out evaluation data in training output. Default false. */\n allowHeldOutTrainingData?: boolean\n /** Require quality to be strictly greater than this value. Default 0. */\n minimumQualityExclusive?: number\n /**\n * Explicit split selection, replacing the default trainable-split rule.\n * Use this only when producing a deliberately named non-training slice.\n */\n splitFilter?: RolloutSplit[]\n}\n\nexport interface GrpoLookups\n extends Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'> {\n /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the assistant completion text for a rollout. */\n completionOf: (runId: string) => string | Promise<string>\n}\n\nexport interface GrpoExportRow {\n prompt: string\n completions: string[]\n rewards: number[]\n /** runIds in the same order as `completions[]` for traceability. */\n runIds: string[]\n meta?: Record<string, unknown>\n}\n\n/**\n * Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —\n * one row per scenario, with one completion per rollout on that scenario.\n * A scenario with fewer than two rewarded completions emits no row because a\n * group of one has no relative baseline.\n *\n * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages\n * within a group of completions for the same prompt; this is the\n * canonical input format. That relative baseline is exactly why the gate has\n * to hold here: one gamed sibling exporting at full reward shifts the advantage\n * of every honest run beside it.\n *\n * On the line path a realness-gated line stays in its group at reward 0 rather\n * than being dropped. 0 is the honest label for a faked success and is usable\n * signal; removing the line would also move the group's baseline, just in the\n * other direction. (SFT differs — see `toSftRows`.)\n */\nexport async function toGrpoRows(\n lines: MintedRolloutLine[],\n lookups: GrpoLookups,\n): Promise<GrpoExportRow[]> {\n return grpoRowsFromLines(lines, lookups)\n}\n\nasync function grpoRowsFromLines(\n lines: MintedRolloutLine[],\n lookups: GrpoLookups,\n): Promise<GrpoExportRow[]> {\n const grouped = new Map<string, MintedRolloutLine[]>()\n for (const line of lines) {\n if (!isSelectedSplit(line, lookups)) continue\n const arr = grouped.get(line.task.instance_id) ?? []\n arr.push(line)\n grouped.set(line.task.instance_id, arr)\n }\n\n const rows: GrpoExportRow[] = []\n for (const [scenarioId, group] of grouped.entries()) {\n if (group.length === 0) continue\n const scored: Array<{ line: MintedRolloutLine; reward: number }> = []\n for (const line of group) {\n const reward = trainableLineReward(line)\n if (reward === null) continue\n scored.push({ line, reward })\n }\n // GRPO's advantage is relative to the group mean, and a single completion\n // has no baseline.\n if (scored.length < 2) continue\n const prompts = await Promise.all(\n scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id))),\n )\n const prompt = prompts[0]!\n if (prompts.some((value) => value !== prompt)) {\n throw new Error(\n `toGrpoRows: scenario \"${scenarioId}\" resolves to different prompt text within one group`,\n )\n }\n const completions = await Promise.all(\n scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id))),\n )\n const rewards = scored.map(({ reward }) => reward)\n const runIds = scored.map(({ line }) => line.run_id)\n rows.push({\n prompt,\n completions,\n rewards,\n runIds,\n meta: {\n scenarioId,\n n: completions.length,\n meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length,\n },\n })\n }\n return rows\n}\n\nexport function toGrpoJsonl(rows: GrpoExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── SFT ──────────────────────────────────────────────────────────────────\n\nexport interface SftLookups extends TrainingLineSelectionOptions {\n /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the assistant completion text for a rollout. */\n completionOf: (runId: string) => string | Promise<string>\n /** Optional system message. Default omits. */\n systemOf?: (line: MintedRolloutLine) => string | null | undefined\n /** Extra filter on top of the realness gate (e.g., low score, failed cases). */\n include?: (line: MintedRolloutLine) => boolean\n /** Include held-out lines under the default split rule. Default false. */\n allowHeldOutTrainingData?: boolean\n}\n\nexport interface SftExportRow {\n messages: Array<{ role: 'system' | 'user' | 'assistant'; content: string }>\n meta?: Record<string, unknown>\n}\n\n/**\n * Convert rollout lines into Hugging Face / OpenAI / Anthropic-style\n * conversational SFT rows. By default every qualifying line becomes one row;\n * pass `include` to filter further (e.g., keep only `reward >= 0.8` for\n * rejection-sampling SFT).\n *\n * Realness-gated lines are dropped outright, not zeroed. SFT is imitation\n * learning: unlike GRPO, where a 0 reward teaches \"this trajectory was bad\",\n * every row here is a target to copy, so a gamed trajectory must not be in the\n * file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.\n *\n * The exporter is fail-closed on the split, same rule as\n * `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by\n * default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and\n * `canary` never pass the default rule. A non-training bundle that wants an\n * explicit slice (e.g. a holdout-only eval bundle) names it with\n * `splitFilter: ['holdout']` — explicit selection replaces the default rule.\n */\nexport async function toSftRows(\n lines: MintedRolloutLine[],\n lookups: SftLookups,\n): Promise<SftExportRow[]> {\n return sftRowsFromLines(lines, lookups)\n}\n\nasync function sftRowsFromLines(\n lines: MintedRolloutLine[],\n lookups: SftLookups,\n): Promise<SftExportRow[]> {\n const include = lookups.include ?? (() => true)\n const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0\n if (!Number.isFinite(minimumQualityExclusive)) {\n throw new Error('minimumQualityExclusive must be finite')\n }\n const rows: SftExportRow[] = []\n for (const line of lines) {\n // Checked BEFORE the drop, so this path fails loud on an impossible line\n // exactly like `rollout/exporters.toSftRows` does rather than quietly\n // filtering it as if it were an ordinary gated row.\n assertRewardGate(line, 'SFT export')\n if (isLineRealnessGated(line)) continue\n if (!isSelectedSplit(line, lookups)) continue\n const score = trainableLineReward(line)\n if (score === null || score <= minimumQualityExclusive) continue\n if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {\n continue\n }\n if (!include(line)) continue\n const system = lookups.systemOf?.(line)\n const [prompt, completion] = await Promise.all([\n Promise.resolve(lookups.promptOf(line.run_id)),\n Promise.resolve(lookups.completionOf(line.run_id)),\n ])\n const messages: SftExportRow['messages'] = []\n if (system) messages.push({ role: 'system', content: system })\n messages.push({ role: 'user', content: prompt })\n messages.push({ role: 'assistant', content: completion })\n rows.push({\n messages,\n meta: {\n runId: line.run_id,\n candidateId: line.candidate_id ?? null,\n scenarioId: line.task.instance_id,\n score,\n model: line.policy.model,\n },\n })\n }\n return rows\n}\n\nexport function toSftJsonl(rows: SftExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── PRM ──────────────────────────────────────────────────────────────────\n\nexport interface PrmLookups {\n /** Resolve the prompt text for a run. */\n promptOf: (runId: string) => string | Promise<string>\n /** Resolve the trajectory step text for a (runId, spanId) pair. */\n stepTextOf: (runId: string, spanId: string) => string | Promise<string>\n /** Optional: sequence of prefix span ids leading up to the divergence. */\n prefixOf?: (runId: string, prefixStepIndex: number) => string[] | Promise<string[]>\n}\n\nexport interface PrmExportRow {\n prompt: string\n /** Span ids for the steps before divergence — caller resolves text via `stepTextOf`. */\n prefixSpanIds: string[]\n prefixStepText: string[]\n chosenStep: string\n rejectedStep: string\n chosenReward: number\n rejectedReward: number\n marginScore: number\n meta?: Record<string, unknown>\n}\n\nexport interface PrmLineContext extends RolloutLineContext {\n /**\n * The `maxSteps` cap the lines were minted with, if any.\n *\n * `mintRolloutRows` drops the MIDDLE of an over-long trajectory and leaves no\n * marker on the line, so a capped trajectory is indistinguishable from a\n * short one. Declaring the cap lets this exporter refuse any line sitting at\n * it — a process-reward model trained on a trajectory with a hole in it\n * learns credit assignment that never happened.\n */\n mintedWithMaxSteps?: number\n}\n\n/**\n * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`\n * callback resolves span text from the consumer's trace store.\n *\n * Every referenced run is checked against its minted line before any row is\n * emitted, and the export FAILS LOUD on a trajectory that was never fully\n * captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected\n * side is realness-gated are dropped instead: a capture defect is the caller's\n * mint configuration and must be fixed, whereas a gamed run is exactly the\n * condition the gate exists to filter.\n *\n * `context` is REQUIRED. A two-argument call used to be accepted and produced\n * rows with no gate applied at all — a `PrmTrainingTriple` carries a bare\n * `chosenReward` number and nothing that says which run it came from is honest,\n * so with no lines this exporter has no way to learn that its chosen step is a\n * step from a run that faked its success. It now throws: fail closed, because\n * the alternative is a process-reward model taught to prefer the gaming move at\n * the exact step the gaming happened.\n */\nexport async function toPrmRows(\n triples: PrmTrainingTriple[],\n lookups: PrmLookups,\n context: PrmLineContext,\n): Promise<PrmExportRow[]> {\n const admitted = admitPrmTriples(triples, context)\n const rows: PrmExportRow[] = []\n for (const t of admitted) {\n const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId))\n const prefixSpanIds = lookups.prefixOf\n ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex))\n : []\n const prefixStepText: string[] = []\n for (const spanId of prefixSpanIds) {\n prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)))\n }\n const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId))\n const rejectedStep = await Promise.resolve(\n lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId),\n )\n rows.push({\n prompt,\n prefixSpanIds,\n prefixStepText,\n chosenStep,\n rejectedStep,\n chosenReward: t.chosenReward,\n rejectedReward: t.rejectedReward,\n marginScore: t.marginScore,\n meta: {\n prefixRunId: t.prefixRunId,\n rejectedRunId: t.rejectedRunId,\n prefixStepIndex: t.prefixStepIndex,\n },\n })\n }\n return rows\n}\n\n/**\n * Refuse to build a process-reward row from a trajectory we do not fully have.\n *\n * PRM training assigns credit step by step, so a missing or silently shortened\n * step list is not degraded data — it is data about a trajectory that never\n * existed. Every condition below throws rather than filters, because each one\n * means the CALLER's capture or mint configuration is wrong.\n */\nfunction assertPrmTrainableLine(line: MintedRolloutLine, mintedWithMaxSteps?: number): void {\n const id = line.rollout_id\n if (line.provenance.gap !== undefined) {\n throw new Error(\n `PRM export: rollout ${id} is a gap line (${line.provenance.gap}) — refusing to build a process-reward row from a trajectory that was never captured`,\n )\n }\n if (line.steps === undefined || line.steps.length === 0) {\n throw new Error(\n `PRM export: rollout ${id} carries no steps — refusing to build a process-reward row with no trajectory`,\n )\n }\n if (line.outcome.is_truncated) {\n throw new Error(\n `PRM export: rollout ${id} is marked truncated — refusing to assign step-level credit over a partial trajectory`,\n )\n }\n if (mintedWithMaxSteps !== undefined && line.steps.length >= mintedWithMaxSteps) {\n throw new Error(\n `PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} — its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`,\n )\n }\n}\n\nconst PRM_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'PRM export',\n contextType: 'PrmLineContext',\n because:\n 'without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one.',\n}\n\n/**\n * Validate every referenced line up front (fail loud, before a single row is\n * written) and then drop the triples whose evidence is realness-gated.\n *\n * The gate half is `admitUngatedByInvocation`, shared with `toDpoRows` and\n * `stepRewardsToJsonl`; only the trajectory-completeness rules are PRM's own.\n */\nfunction admitPrmTriples(\n triples: PrmTrainingTriple[],\n context: PrmLineContext,\n): PrmTrainingTriple[] {\n return admitUngatedByInvocation(\n triples,\n (t) => [t.prefixRunId, t.rejectedRunId],\n context,\n PRM_CONTEXT_REQUIREMENT,\n (line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps),\n )\n}\n\nexport function toPrmJsonl(rows: PrmExportRow[]): string {\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── Step rewards (for value-function regression) ─────────────────────────\n\nexport interface StepRewardJsonlRow {\n runId: string\n spanId: string\n stepIndex: number\n reward: number\n determinism: 'deterministic' | 'probabilistic'\n weight: number\n}\n\nconst STEP_REWARD_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'step-reward export',\n contextType: 'RolloutLineContext',\n because:\n 'a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success — so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere.',\n}\n\n/**\n * Step-level reward rows as JSONL.\n *\n * `context` is REQUIRED for the same reason it is on `toDpoRows` and\n * `toPrmRows`: this is a line-less input carrying a reward number. Steps\n * belonging to a realness-gated run are dropped rather than zeroed — a\n * per-step reward of 0 across a whole trajectory is a claim that every step was\n * bad, which is a different (and false) statement from \"this run's success was\n * fabricated, so its step-level credit assignment is meaningless\".\n */\nexport function stepRewardsToJsonl(stepRewards: StepReward[], context: RolloutLineContext): string {\n const admitted = admitUngatedByInvocation(\n stepRewards,\n (s) => [s.runId],\n context,\n STEP_REWARD_CONTEXT_REQUIREMENT,\n )\n const rows: StepRewardJsonlRow[] = admitted.map((s) => ({\n runId: s.runId,\n spanId: s.spanId,\n stepIndex: s.stepIndex,\n reward: s.reward,\n determinism: s.determinism,\n weight: s.weight ?? 1,\n }))\n return rows.map((r) => JSON.stringify(r)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\nfunction isSelectedSplit(\n line: MintedRolloutLine,\n options: Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'>,\n): boolean {\n if (options.splitFilter !== undefined) return options.splitFilter.includes(line.task.split)\n return isSplitEligible(line, options)\n}\n","/**\n * RL dataset packaging + datasheet — the publishable, sellable bundle.\n *\n * The format exporters (`toGrpoRows` / `toSftRows` / `toDpoRows`) already\n * produce trainer-ready shapes (prime-rl GRPO, TRL DPO, conversational SFT).\n * What turns that into a dataset someone can PUBLISH or BUY is the provenance\n * + a datasheet: which models produced it, which prompt/agent versions, how the\n * reward was derived (deterministic verifiable vs probabilistic judge — the\n * credibility axis a buyer checks first), the split discipline, the reward\n * distribution, the quality gates, the license, and the intended/out-of-scope\n * uses. This module computes those facts from `MintedRolloutLine[]` and renders a\n * \"Datasheet for Datasets\" (Gebru et al. 2018) card alongside the format files.\n *\n * It composes the existing `rl/exporters` — it does not reimplement any trainer\n * format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization\n * with per-token loss masks) is a downstream Python stage that consumes the\n * `messages`/`completions` this bundle emits.\n *\n * Input discipline: the bundle is built from `MintedRolloutLine[]`. The datasheet's\n * reward distribution ships INSIDE the published artifact, so it has to be\n * derived from exactly the same gated number as the rows it describes —\n * otherwise the provenance a buyer checks first is a lie. Taking the same\n * gated input as the exporters is what guarantees that.\n */\n\nimport type { MintedRolloutLine, RolloutSplit } from '../rollout/schema'\nimport {\n type DpoLookups,\n type GrpoLookups,\n type SftLookups,\n toDpoJsonl,\n toDpoRows,\n toGrpoJsonl,\n toGrpoRows,\n toSftJsonl,\n toSftRows,\n} from './exporters'\nimport type { PreferenceTriple } from './preferences'\nimport { trainableLineReward } from './rollout-input'\n\nexport type RewardKind = 'deterministic' | 'probabilistic' | 'mixed'\nconst DATASET_FORMATS = ['grpo', 'sft', 'dpo'] as const\nexport type DatasetFormat = (typeof DATASET_FORMATS)[number]\n\nconst DATASET_FORMAT_SET = new Set<unknown>(DATASET_FORMATS)\n\nexport function validateDatasetFormats(value: unknown): DatasetFormat[] {\n if (!Array.isArray(value) || value.length === 0) {\n throw new Error('buildRlDataset: formats must contain at least one of: grpo, sft, dpo')\n }\n\n const formats: DatasetFormat[] = []\n const seen = new Set<DatasetFormat>()\n for (const format of value) {\n if (!DATASET_FORMAT_SET.has(format)) {\n throw new Error(\n `buildRlDataset: unsupported format ${JSON.stringify(format)}; expected exactly one of: grpo, sft, dpo`,\n )\n }\n const datasetFormat = format as DatasetFormat\n if (seen.has(datasetFormat)) {\n throw new Error(\n `buildRlDataset: duplicate format ${JSON.stringify(datasetFormat)}; each format may be requested once`,\n )\n }\n seen.add(datasetFormat)\n formats.push(datasetFormat)\n }\n return formats\n}\n\n/** Caller-declared context — the qualitative half of the datasheet that can't\n * be computed from records. */\nexport interface RlDatasetConfig {\n name: string\n version: string\n /** Product/task domain, e.g. 'legal-m&a', 'tax-1040'. */\n domain: string\n /** SPDX id or a named commercial license. Required — an unlicensed dataset\n * cannot be published or sold. */\n license: string\n /** How the reward was produced. `kind: 'deterministic'` (a test/schema/XPath\n * decided it) is the credibility signal; 'probabilistic' = LLM-judge. */\n reward: { kind: RewardKind; source: string; description: string }\n intendedUse: string\n outOfScope: string\n limitations: string\n /** ISO timestamp — passed in (the substrate forbids Date.now()). */\n createdAtIso: string\n /** Default: ['sft']. GRPO must be requested for multi-completion groups. */\n formats?: DatasetFormat[]\n /** Quality gates already run, recorded on the card for the buyer. */\n qualityGates?: {\n contaminationProbe?: 'passed' | 'failed' | 'not-run'\n dedup?: boolean\n verifiableRewardFilter?: boolean\n }\n}\n\nexport interface RewardStats {\n n: number\n mean: number | null\n median: number | null\n min: number | null\n max: number | null\n std: number | null\n}\n\nexport interface RlDatasetStats {\n records: number\n /** Rollouts carrying an explicit task-quality score. */\n scoredRecords: number\n /** Rollout count per split. */\n splits: Record<RolloutSplit, number>\n reward: RewardStats\n /** Distinct snapshot-pinned models that produced the trajectories. */\n models: string[]\n /** Distinct effective-prompt hashes (the agent profile/prompt versions). */\n promptHashes: string[]\n commitShas: string[]\n totalTokens: { input: number; output: number }\n totalCostUsd: number\n /**\n * Rollouts whose USD cost was never captured (`cost.usd === null`). When\n * non-zero, `totalCostUsd` is a floor, not the bill — a published dataset\n * must not present an unbilled run as a $0 one.\n */\n rolloutsWithoutCost: number\n}\n\nexport interface RlDatasetManifest extends RlDatasetConfig {\n formats: DatasetFormat[]\n rowCounts: Partial<Record<DatasetFormat, number>>\n stats: RlDatasetStats\n}\n\nexport interface RlDatasetBundle {\n manifest: RlDatasetManifest\n /** Relative filename -> contents. Write these to a directory to publish. */\n files: Record<string, string>\n}\n\nfunction distinct(xs: Array<string | null | undefined>): string[] {\n return [...new Set(xs.filter((x): x is string => typeof x === 'string' && x.length > 0))].sort()\n}\n\nfunction computeRewardStats(values: number[]): RewardStats {\n if (values.length === 0) {\n return { n: 0, mean: null, median: null, min: null, max: null, std: null }\n }\n const sorted = [...values].sort((a, b) => a - b)\n const n = sorted.length\n const mean = sorted.reduce((s, x) => s + x, 0) / n\n const mid = Math.floor(n / 2)\n const median = n % 2 === 0 ? (sorted[mid - 1]! + sorted[mid]!) / 2 : sorted[mid]!\n const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n\n return { n, mean, median, min: sorted[0]!, max: sorted[n - 1]!, std: Math.sqrt(variance) }\n}\n\nfunction computeStatsFromLines(lines: MintedRolloutLine[]): RlDatasetStats {\n const splits: Record<RolloutSplit, number> = { search: 0, dev: 0, holdout: 0, canary: 0 }\n let inTok = 0\n let outTok = 0\n let cost = 0\n let rolloutsWithoutCost = 0\n const rewards: number[] = []\n for (const line of lines) {\n splits[line.task.split] += 1\n inTok += line.cost.tokens_in ?? 0\n outTok += line.cost.tokens_out ?? 0\n if (line.cost.usd === null) rolloutsWithoutCost++\n else cost += line.cost.usd\n const rw = trainableLineReward(line)\n if (rw !== null) rewards.push(rw)\n }\n return {\n records: lines.length,\n scoredRecords: rewards.length,\n splits,\n reward: computeRewardStats(rewards),\n models: distinct(lines.map((l) => l.policy.model)),\n promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),\n commitShas: distinct(lines.map((l) => l.policy.profile_commit)),\n totalTokens: { input: inTok, output: outTok },\n totalCostUsd: cost,\n rolloutsWithoutCost,\n }\n}\n\n/**\n * Package graded rollout lines into a publishable RL dataset bundle: the\n * trainer-format JSONL files + a manifest + a datasheet. DPO requires\n * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from\n * the lines directly via the supplied lookups. Throws on an empty corpus —\n * an empty dataset must never be published.\n */\nexport async function buildRlDataset(\n lines: MintedRolloutLine[],\n lookups: GrpoLookups & SftLookups,\n config: RlDatasetConfig,\n preferences?: { triples: PreferenceTriple[]; lookups: DpoLookups },\n): Promise<RlDatasetBundle> {\n if (lines.length === 0) {\n throw new Error('buildRlDataset: no rollout lines — refusing to package an empty dataset')\n }\n const formats = validateDatasetFormats(config.formats === undefined ? ['sft'] : config.formats)\n const files: Record<string, string> = {}\n const rowCounts: Partial<Record<DatasetFormat, number>> = {}\n\n if (formats.includes('grpo')) {\n const rows = await toGrpoRows(lines, lookups)\n requireRows('grpo', rows.length)\n files['train.grpo.jsonl'] = toGrpoJsonl(rows)\n rowCounts.grpo = rows.length\n }\n if (formats.includes('sft')) {\n const rows = await toSftRows(lines, lookups)\n requireRows('sft', rows.length)\n files['train.sft.jsonl'] = toSftJsonl(rows)\n rowCounts.sft = rows.length\n }\n if (formats.includes('dpo')) {\n if (!preferences) {\n throw new Error(\"buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)\")\n }\n const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines })\n requireRows('dpo', rows.length)\n files['train.dpo.jsonl'] = toDpoJsonl(rows)\n rowCounts.dpo = rows.length\n }\n if (!Object.keys(files).some((name) => name.startsWith('train.') && name.endsWith('.jsonl'))) {\n throw new Error('buildRlDataset: no trainer file was emitted')\n }\n\n const manifest: RlDatasetManifest = {\n ...config,\n formats,\n rowCounts,\n stats: computeStatsFromLines(lines),\n }\n files['manifest.json'] = `${JSON.stringify(manifest, null, 2)}\\n`\n files['DATASHEET.md'] = datasheetToMarkdown(manifest)\n return { manifest, files }\n}\n\nfunction requireRows(format: DatasetFormat, rows: number): void {\n if (rows === 0) {\n throw new Error(`buildRlDataset: requested '${format}' format produced no trainable rows`)\n }\n}\n\nfunction pct(x: number): string {\n return `${(x * 100).toFixed(1)}%`\n}\n\nfunction stat(value: number | null): string {\n return value === null ? 'n/a' : value.toFixed(3)\n}\n\n/** Render the \"Datasheet for Datasets\" card that a buyer reads. */\nexport function datasheetToMarkdown(m: RlDatasetManifest): string {\n const s = m.stats\n const total = s.records || 1\n const splitLines = (['search', 'dev', 'holdout', 'canary'] as RolloutSplit[])\n .map((k) => ` - \\`${k}\\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`)\n .join('\\n')\n const costNote =\n s.rolloutsWithoutCost > 0\n ? ` (floor — ${s.rolloutsWithoutCost} rollout(s) never captured a cost)`\n : ''\n const deterministic = m.reward.kind === 'deterministic'\n return [\n `# Dataset: ${m.name} \\`v${m.version}\\``,\n '',\n `**Domain:** ${m.domain} | **Created:** ${m.createdAtIso} | **License:** ${m.license}`,\n '',\n '## Reward provenance',\n `- **Kind:** ${m.reward.kind}${deterministic ? ' (decidable, not judge noise)' : ''}`,\n `- **Source:** ${m.reward.source}`,\n `- **Description:** ${m.reward.description}`,\n '',\n '## Composition',\n `- **Records (trajectories):** ${s.records}`,\n `- **Scored records:** ${s.scoredRecords}`,\n `- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(', ')}`,\n '- **Splits:**',\n splitLines,\n '',\n '## Reward distribution',\n `- n=${s.reward.n} | mean=${stat(s.reward.mean)} | median=${stat(s.reward.median)} | min=${stat(s.reward.min)} | max=${stat(s.reward.max)} | std=${stat(s.reward.std)}`,\n '',\n '## Provenance',\n `- **Models:** ${s.models.join(', ')}`,\n `- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,\n `- **Commits:** ${s.commitShas.join(', ')}`,\n `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,\n '',\n '## Quality gates',\n `- Contamination probe: ${m.qualityGates?.contaminationProbe ?? 'not-run'}`,\n `- Dedup: ${m.qualityGates?.dedup ? 'yes' : 'no'} | Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? 'yes' : 'no'}`,\n '',\n '## Recommended uses',\n m.intendedUse,\n '',\n '## Out of scope',\n m.outOfScope,\n '',\n '## Limitations',\n m.limitations,\n '',\n '## Token rendering',\n 'For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns. See `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.',\n '',\n ].join('\\n')\n}\n","/**\n * RL corpus — the durable, append-only accumulation of graded RunRecords that\n * every eval run deposits BY DEFAULT.\n *\n * The dataset is the free exhaust of the normal eval process: we run evals\n * constantly to get an agent production-ready, and those runs already produce\n * graded trajectories. Instead of writing them to an ephemeral run dir and\n * throwing them away, `appendToCorpus` accumulates them into a durable corpus;\n * `buildDatasetFromCorpus` later harvests the whole corpus into a publishable\n * bundle. No separate data-collection campaign — the data accrues from work we\n * do anyway. This is the \"best things for free by our process\" layer.\n *\n * Trajectory text rides on the record as top-level `prompt` / `completion`\n * (what the eval harnesses capture; the RunRecord validator ignores the extra\n * keys). The harvest reads them directly — no trace store round-trip needed.\n */\n\nimport { appendFileSync, existsSync, mkdirSync, readFileSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { mintRolloutRows } from '../rollout/mint'\nimport { trainingScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { InMemoryTraceStore } from '../trace/store'\nimport { buildRlDataset, type RlDatasetBundle, type RlDatasetConfig } from './dataset'\n\n/** A corpus record is a RunRecord carrying the trajectory text the harness\n * captured. `prompt`/`completion` are top-level (the validator ignores extras). */\nexport type CorpusRecord = RunRecord & { prompt?: string; completion?: string }\n\nexport interface CorpusAppendResult {\n appended: number\n /** Skipped because a record with the same runId was already in the corpus\n * (idempotent appends — NOT re-run collapsing; re-runs get fresh runIds). */\n skipped: number\n total: number\n}\n\n/**\n * Append graded records to the corpus (append-only JSONL). Deduplicates by\n * `runId` against what's already on disk so re-running the same harness is\n * idempotent. Creates the file and parent dir. This is the call every eval\n * harness makes by default after producing its records.\n */\nexport function appendToCorpus(records: CorpusRecord[], corpusPath: string): CorpusAppendResult {\n mkdirSync(dirname(corpusPath), { recursive: true })\n const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : []\n const seen = new Set(existing.map((r) => r.runId))\n const lines: string[] = []\n let appended = 0\n let skipped = 0\n for (const r of records) {\n if (seen.has(r.runId)) {\n skipped++\n continue\n }\n seen.add(r.runId)\n lines.push(JSON.stringify(r))\n appended++\n }\n if (lines.length > 0) appendFileSync(corpusPath, `${lines.join('\\n')}\\n`)\n return { appended, skipped, total: existing.length + appended }\n}\n\n/** Read the full corpus. Returns [] if the corpus does not exist yet. */\nexport function readCorpus(corpusPath: string): CorpusRecord[] {\n if (!existsSync(corpusPath)) return []\n const out: CorpusRecord[] = []\n for (const line of readFileSync(corpusPath, 'utf8').split('\\n')) {\n if (line.trim()) out.push(JSON.parse(line) as CorpusRecord)\n }\n return out\n}\n\n/**\n * The harvest's score reader is GATED: a gamed run reads 0, so it cannot buy\n * its way past `minScore` into the published bundle with its claimed score.\n * `null` = unscored (a labeled gap, dropped before packaging, never a 0).\n */\nfunction rewardOf(r: CorpusRecord): number | null {\n const v = trainingScore(r)\n return typeof v === 'number' && Number.isFinite(v) ? v : null\n}\n\nexport interface HarvestOptions {\n /** Keep only records scoring >= this (rejection-sampling for SFT). */\n minScore?: number\n /** Keep only these source splits. Held-out rows still require the explicit override below. */\n splits?: RunRecord['splitTag'][]\n /** Permit held-out rows in training files. Default false. */\n allowHeldOutTrainingData?: boolean\n}\n\n/**\n * Harvest the accumulated corpus into a publishable RL dataset bundle. Reads\n * trajectory text from each record's top-level `prompt`/`completion`; records\n * missing either are excluded (a graded score with no trajectory can't train).\n * Optionally filters by score / split. Throws (via buildRlDataset) if nothing\n * survives — an empty dataset must never be published.\n *\n * `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run\n * cannot buy its way into the published bundle with its claimed score —\n * `minScore` is exactly the door a reward-hacked run would otherwise clear for\n * SFT. Unscored records are dropped before packaging: a missing label is not a\n * zero, and it is not publishable either.\n */\nexport async function buildDatasetFromCorpus(\n corpusPath: string,\n config: RlDatasetConfig,\n opts: HarvestOptions = {},\n): Promise<RlDatasetBundle> {\n let records = readCorpus(corpusPath).filter(\n (r) => typeof r.prompt === 'string' && typeof r.completion === 'string',\n )\n if (opts.splits) records = records.filter((r) => opts.splits!.includes(r.splitTag))\n records = records.filter((r) => rewardOf(r) !== null)\n if (opts.minScore != null) {\n records = records.filter((r) => {\n const reward = rewardOf(r)\n return reward !== null && reward >= opts.minScore!\n })\n }\n\n const text = new Map(\n records.map((r) => [r.runId, { prompt: r.prompt!, completion: r.completion! }]),\n )\n const lookups = {\n promptOf: (id: string) => text.get(id)?.prompt ?? '',\n completionOf: (id: string) => text.get(id)?.completion ?? '',\n allowHeldOutTrainingData: opts.allowHeldOutTrainingData,\n }\n const { rows } = await mintRolloutRows(records, new InMemoryTraceStore())\n return buildRlDataset(rows, lookups, config)\n}\n","/**\n * Off-policy evaluation primitives.\n *\n * Standard inverse-probability-weighted (IPS), self-normalized\n * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the\n * value of a *target* policy given trajectories collected under a\n * *behavior* policy. This is the canonical RL eval task: \"we have last\n * week's runs, we changed the policy — how would the new one do without\n * re-running?\"\n *\n * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan\n * & Joachims 2015 for SNIPS) but the *application* to LLM-agent\n * evaluation needs care:\n *\n * - The \"policy\" is the (prompt, tool config, model snapshot) triple.\n * Two policies have the same probability over an action *iff* their\n * LLM call would emit the same token with the same probability —\n * which is generally unknowable without the model log-probs.\n * - For LLM agents, propensity scores must be supplied by the caller\n * (logged in the trace, recovered from token log-probs, or estimated\n * via a learned propensity model). We do NOT estimate propensity here.\n * - Doubly-robust requires two outputs from a Q-function: its prediction\n * for the logged action and its expectation under the target policy.\n * Consumers compute these with a tabular estimate, regression fit, or\n * learned reward model before constructing the trajectories.\n *\n * Bias / variance tradeoffs:\n * - IPS: unbiased; high variance for small overlap, infinite variance\n * when target has support outside behavior.\n * - SNIPS: lower variance, slight bias; usually preferred in practice.\n * - DR: doubly-robust — unbiased if either propensity OR Q-function is\n * correct. Lowest practical variance when Q is decent. Use this.\n *\n * Caveat the panel will land: on the LLM-agent setting, propensity scores\n * recovered from token log-probs are noisy, the action space is enormous,\n * and overlap is often poor. These estimators are useful but not magic;\n * complement with `replayCampaign` (exact replay where the request hashes\n * match) for high-confidence answers and OPE for the gap.\n */\n\nimport { ValidationError } from '../errors'\n\nexport interface OffPolicyTrajectory {\n /** Stable id, for traceability through the dataset. */\n runId: string\n /** Reward observed under the behavior policy (the realized outcome). */\n reward: number\n /**\n * Behavior-policy probability of the action that was taken. For LLM\n * agents this is typically `exp(sum(token_log_probs))` over the chosen\n * trajectory. Must be in (0, 1].\n */\n behaviorProb: number\n /**\n * Target-policy probability of the same action. For replay-style\n * counterfactual evaluation this is what the *new* policy would have\n * assigned to the *old* trajectory. Must be in [0, 1].\n */\n targetProb: number\n /**\n * Model-based reward prediction for the action selected by the behavior\n * policy: `Q_hat(context, loggedAction)`. Supply this together with\n * `vHatTarget` for contextual-bandit doubly-robust estimation.\n */\n qHatChosen?: number | null\n /**\n * Expected model-based reward under the target policy:\n * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.\n * Supply this together with `qHatChosen`. For an honest evaluation, both\n * values must come from a model cross-fitted or trained outside this row.\n */\n vHatTarget?: number | null\n}\n\nexport interface OffPolicyContributionCounts {\n /** Contributions using the contextual-bandit doubly-robust formula. */\n dr: number\n /** Contributions using exact IPS because no reward-model estimate was supplied. */\n ipsFallback: number\n}\n\nexport interface OffPolicyEstimate {\n /** Estimated value of the target policy. */\n value: number\n /** Standard error of the estimate. */\n standardError: number\n /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */\n effectiveSampleSize: number\n /** Number of trajectories used. */\n n: number\n /**\n * Diagnostic: maximum importance weight observed. Large values (>>10x\n * mean) are a red flag — variance is dominated by a few outliers.\n */\n maxImportanceWeight: number\n /** Populated by `doublyRobust` to expose which formula each row used. */\n contributionCounts?: OffPolicyContributionCounts\n}\n\nexport interface OffPolicyOptions {\n /**\n * Cap importance weights at this value (Ionides 2008 truncated IS) to\n * trade unbiasedness for variance reduction. Default `Infinity` (no cap).\n * Set e.g. `10` for stable estimates when the policies are close.\n */\n weightCap?: number\n /** Reward clipping range. Default `[0, 1]`. */\n rewardClip?: { low: number; high: number }\n}\n\n/**\n * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator\n * of E[reward under target policy]. Variance scales with the spread of\n * target/behavior ratios.\n */\nexport function inverseProbabilityWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n\n if (trajectories.length === 0) {\n return zeroEstimate()\n }\n\n const weights: number[] = []\n const weightedRewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `inverseProbabilityWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n weights.push(w)\n weightedRewards.push(w * r)\n if (w > maxW) maxW = w\n }\n const n = weights.length\n const value = weightedRewards.reduce((s, x) => s + x, 0) / n\n const variance = weightedRewards.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at\n * the cost of small bias (vanishing as N grows). The right default for\n * LLM-agent evaluation where overlap is often poor.\n */\nexport function selfNormalizedImportanceWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) return zeroEstimate()\n\n const weights: number[] = []\n const rewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `selfNormalizedImportanceWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n weights.push(w)\n rewards.push(clamp(t.reward, clip.low, clip.high))\n if (w > maxW) maxW = w\n }\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumWR = weights.reduce((s, w, i) => s + w * rewards[i]!, 0)\n const value = sumW === 0 ? 0 : sumWR / sumW\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n // Influence-function-based SE for SNIPS (Owen 2013, Ch. 9).\n const phi = weights.map((w, i) => w * (rewards[i]! - value))\n const variance = phi.reduce((s, x) => s + x * x, 0) / Math.max(1, sumW * sumW)\n return {\n value,\n standardError: Math.sqrt(variance),\n effectiveSampleSize: effN,\n n: trajectories.length,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).\n *\n * V_DR = (1/N) * sum_i [ v_hat_target_i\n * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]\n *\n * Unbiased if EITHER:\n * - the importance ratios are correct (IPS-style validity), OR\n * - the Q-hat function is correct (model-based validity).\n *\n * In practice both are imperfect, but the residual bias is the *product*\n * of both errors — much smaller than either alone. This is why DR is the\n * default in production OPE pipelines.\n *\n * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither\n * use the exact IPS contribution. `contributionCounts` makes the mix explicit\n * in the result.\n * Callers must cross-fit the Q-function or train it on independent rows;\n * fitting and evaluating Q on the same outcomes leaks the answer.\n */\nexport function doublyRobust(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) {\n return {\n ...zeroEstimate(),\n contributionCounts: { dr: 0, ipsFallback: 0 },\n }\n }\n\n const contributions: number[] = []\n const contributionCounts: OffPolicyContributionCounts = {\n dr: 0,\n ipsFallback: 0,\n }\n let maxW = 0\n let sumW = 0\n let sumW2 = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(`doublyRobust: behaviorProb must be > 0 (runId=${t.runId})`)\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n const rawQHatChosen = t.qHatChosen\n const rawVHatTarget = t.vHatTarget\n const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== undefined\n const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== undefined\n if (hasQHatChosen !== hasVHatTarget) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`,\n )\n }\n\n if (hasQHatChosen && hasVHatTarget) {\n if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`,\n )\n }\n const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high)\n const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high)\n contributions.push(vHatTarget + w * (r - qHatChosen))\n contributionCounts.dr += 1\n } else {\n contributions.push(w * r)\n contributionCounts.ipsFallback += 1\n }\n if (w > maxW) maxW = w\n sumW += w\n sumW2 += w * w\n }\n const n = contributions.length\n const value = contributions.reduce((s, x) => s + x, 0) / n\n const variance = contributions.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n contributionCounts,\n }\n}\n\n/**\n * Convenience: run all three estimators and return them side-by-side.\n * The recommended diagnostic — agreement across estimators is a much\n * stronger signal than any single one.\n */\nexport function offPolicyEstimateAll(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): { ips: OffPolicyEstimate; snips: OffPolicyEstimate; dr: OffPolicyEstimate } {\n return {\n ips: inverseProbabilityWeighting(trajectories, opts),\n snips: selfNormalizedImportanceWeighting(trajectories, opts),\n dr: doublyRobust(trajectories, opts),\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction zeroEstimate(): OffPolicyEstimate {\n return { value: 0, standardError: 0, effectiveSampleSize: 0, n: 0, maxImportanceWeight: 0 }\n}\n\nfunction clamp(x: number, lo: number, hi: number): number {\n if (!Number.isFinite(x)) return lo\n return Math.max(lo, Math.min(hi, x))\n}\n","/**\n * `PredictiveValidityResearcher` — concrete `Researcher` implementation\n * that drives selection from outcome-anchored predictive validity.\n *\n * Each method:\n *\n * - `inspectFailures(runs)` — synthesizes failure modes from the\n * bottom-quartile of `RunRecord`s on the configured proxy reward.\n * - `proposeChange(failures)` — proposes steering changes that target\n * the rubrics with the lowest predictive validity (decorative ones).\n * Either reduce their weight in the composite, or recalibrate them.\n * - `applyChange(changes, baseline)` — merges the proposed steering\n * into the experiment plan.\n * - `evaluateChange(plan)` — re-runs the predictive-validity check on\n * the post-change runs and reports the delta.\n *\n * The result is a closed loop: the rubric weights drift toward the ones\n * that actually predict deployment outcomes, automatically. Pair with\n * `runRLCampaign` for the full auto-research story.\n */\n\nimport type { GateDecision, SplitCoverage } from '../held-out-gate'\nimport type { OutcomeStore } from '../meta-eval/outcome-store'\nimport {\n type RubricPredictiveValidityReport,\n rubricPredictiveValidity,\n} from '../meta-eval/rubric-predictive-validity'\nimport type {\n ExperimentPlan,\n ExperimentResult,\n FailureMode,\n Researcher,\n SteeringChange,\n} from '../researcher'\nimport { type RunRecord, runTaskScore } from '../run-record'\n\nexport interface PredictiveValidityResearcherOptions {\n outcomes: OutcomeStore\n outcomeMetrics: string[]\n /** Score threshold below which a run counts as a \"failure.\" Default 0.5. */\n failureThreshold?: number\n /** Spearman bucket below which a rubric is \"decorative.\" Default 0.4. */\n decorativeThreshold?: number\n /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */\n steeringNamespace?: string\n /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */\n rubrics?: string[]\n /**\n * Snapshot stash hook — called with the most recent predictive-validity\n * report. Useful when a downstream system wants to log rubric drift over\n * time. Default no-op.\n */\n onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>\n}\n\n/**\n * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:\n * rubrics that don't predict deployment outcomes don't earn weight.\n */\nexport class PredictiveValidityResearcher implements Researcher {\n private opts: PredictiveValidityResearcherOptions\n private lastReport: RubricPredictiveValidityReport | null = null\n\n constructor(opts: PredictiveValidityResearcherOptions) {\n this.opts = opts\n }\n\n async inspectFailures(runs: RunRecord[]): Promise<FailureMode[]> {\n const threshold = this.opts.failureThreshold ?? 0.5\n const failures: FailureMode[] = []\n // Ungated: the researcher reports what the runs actually scored. A gamed\n // run scored high and is therefore NOT a low-score failure mode — calling it\n // one here would attribute the wrong failure to the candidate.\n const failingRuns = runs.filter((r) => {\n const score = runTaskScore(r)\n return typeof score === 'number' && score < threshold\n })\n if (failingRuns.length === 0) return failures\n\n // Group failures by candidateId — the researcher's primary handle is\n // \"this candidate is producing low-scoring outputs in this scenario.\"\n const grouped = new Map<string, RunRecord[]>()\n for (const r of failingRuns) {\n const arr = grouped.get(r.candidateId) ?? []\n arr.push(r)\n grouped.set(r.candidateId, arr)\n }\n\n for (const [candidateId, group] of grouped.entries()) {\n const meanScore =\n group.reduce((s, r) => {\n const score = runTaskScore(r)\n if (score === undefined) {\n throw new Error(`failing run ${r.runId} unexpectedly has no task score`)\n }\n return s + score\n }, 0) / group.length\n failures.push({\n code: `low-score-${candidateId}`,\n description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,\n evidence: {\n runIds: group.slice(0, 8).map((r) => r.runId),\n samples: group.length,\n },\n })\n }\n return failures\n }\n\n async proposeChange(failures: FailureMode[]): Promise<SteeringChange[]> {\n if (failures.length === 0) return []\n\n // Without a prior report, return a single \"collect more outcome data\"\n // change — the researcher refuses to reweight rubrics from zero evidence.\n if (this.lastReport === null) {\n return [\n {\n kind: 'threshold',\n payload: { directive: 'researcher.collect-more-outcomes' },\n rationale:\n 'predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists',\n },\n ]\n }\n\n const decorativeThreshold = this.opts.decorativeThreshold ?? 0.4\n const changes: SteeringChange[] = []\n\n for (const ranking of this.lastReport.ranked) {\n if (ranking.verdict === 'load_bearing') continue\n if (Math.abs(ranking.spearman) >= decorativeThreshold) continue\n changes.push({\n kind: 'reviewer_prompt',\n payload: {\n rubric: ranking.rubric,\n action: 'down-weight',\n spearman: ranking.spearman,\n bestOutcome: ranking.bestOutcome,\n },\n rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,\n expectedDelta: -Math.max(0, 0.05 - Math.abs(ranking.spearman)),\n })\n }\n for (const ranking of this.lastReport.ranked.slice(0, 1)) {\n if (ranking.verdict !== 'load_bearing') continue\n changes.push({\n kind: 'reviewer_prompt',\n payload: {\n rubric: ranking.rubric,\n action: 'up-weight',\n spearman: ranking.spearman,\n bestOutcome: ranking.bestOutcome,\n },\n rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,\n expectedDelta: Math.max(0, Math.abs(ranking.spearman) - 0.5) * 0.1,\n })\n }\n return changes\n }\n\n async applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan> {\n // Merge proposed changes into the plan's `changes` array, preserving\n // any changes the baseline already had.\n return {\n ...baseline,\n changes: [...baseline.changes, ...changes],\n }\n }\n\n async evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult> {\n // The researcher contract takes a *plan* and returns a *result* —\n // implementations that only understand re-scoring runs can produce a\n // \"no-op\" gate decision and let the caller drive the actual sweep.\n // Real evaluators (CallbackResearcher) execute the plan; we report.\n const emptyGate: GateDecision = {\n promote: false,\n candidateId: plan.proposedCandidateId,\n baselineId: plan.baselineCandidateId,\n evidence: {\n productiveRuns: 0,\n unpairedCandidateRuns: 0,\n unpairedBaselineRuns: 0,\n medianPairedDelta: null,\n deltaStatistic: 'median_bootstrap',\n decidingDelta: null,\n pairedCI: null,\n pairedPValue: null,\n mcnemar: null,\n binaryScale: null,\n tieFraction: null,\n searchScore: null,\n holdoutScore: null,\n overfitGap: null,\n baselineOverfitGap: null,\n medianCandidateCost: null,\n medianBaselineCost: null,\n realnessGatedRuns: 0,\n // Nothing was dealt, so nothing was answered — this researcher never\n // runs the sweep, it only reports that the caller must.\n holdoutCoverage: emptyCoverage(),\n searchCoverage: emptyCoverage(),\n },\n reason:\n 'predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].',\n rejectionCode: 'few_runs',\n }\n return {\n plan,\n runs: [],\n gateDecision: emptyGate,\n }\n }\n\n /**\n * Run the predictive-validity check explicitly against a fresh RunRecord\n * set. Updates the researcher's cached report so subsequent\n * `proposeChange` calls have evidence to draw from.\n */\n async runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport> {\n const report = await rubricPredictiveValidity({\n runs,\n outcomes: this.opts.outcomes,\n outcomeMetrics: this.opts.outcomeMetrics,\n rubrics: this.opts.rubrics,\n })\n if (this.opts.onReport) await this.opts.onReport(report)\n this.lastReport = report\n return report\n }\n\n /**\n * Force-feed a predictive-validity report into the researcher state —\n * useful when the consumer ran the report out-of-band and wants the\n * researcher's later proposals informed by it.\n */\n setReport(report: RubricPredictiveValidityReport): void {\n this.lastReport = report\n }\n\n getLastReport(): RubricPredictiveValidityReport | null {\n return this.lastReport\n }\n}\n\n/** Coverage of a split that was never dealt any work. */\nfunction emptyCoverage(): SplitCoverage {\n return { dealt: 0, answered: 0, unscoredPairs: 0, candidateOnly: 0, baselineOnly: 0, coverage: 0 }\n}\n","/**\n * Preference dataset extraction from canonical minted rollout lines.\n *\n * Production RLHF / DPO / KTO / SimPO pipelines need preference triples:\n * `(prompt, chosen, rejected)`. The campaign artifact already contains the\n * ingredients — every (variantId, scenarioId, seed) cell is a candidate\n * that ran the same prompt against the same scenario, scored by the same\n * judge — but turning that into a clean preference dataset requires\n * deciding *what counts as a preference*.\n *\n * This module ships three preference-extraction strategies with explicit\n * tradeoffs, plus a unified output type compatible with HuggingFace TRL,\n * Anthropic finetuning JSONL, and OpenAI fine-tuning APIs. The strategies\n * are deliberately not auto-magical — picking the wrong one corrupts the\n * gradient.\n *\n * Strategies:\n *\n * 1. **`paired-by-scenario-and-seed`** — exact-match comparisons. For\n * each scenario × seed pair, compare every (variantA, variantB) on\n * that exact (scenario, seed). Matches scenarios so the comparison\n * isolates variant effects. Highest signal-to-noise; smallest\n * dataset (only matched pairs count).\n *\n * 2. **`paired-by-scenario`** — looser matching. For each scenario,\n * compare every (variantA, variantB) where both have ≥ 1 run on the\n * same scenario. Aggregates across seeds to compute mean scores per\n * (variant, scenario), then forms preferences from the means. More\n * data, lower per-pair signal.\n *\n * 3. **`top-vs-bottom`** — coarsest. Within each scenario, the highest-\n * scoring run is `chosen`, the lowest is `rejected`. Smallest dataset\n * per scenario but biggest score gap per pair. Useful for early\n * bootstrapping when you have few variants.\n *\n * The output `PreferenceTriple` is *agent-eval-canonical* but trivially\n * mappable to TRL's `DPODataset` shape (`prompt`, `chosen`, `rejected`)\n * via the `toTRLFormat` helper, which resolves real prompt/completion text\n * through the same lookups `toDpoRows` takes (`./exporters` carries the\n * richer row with margin + metadata).\n *\n * Input discipline: the function accepts only `MintedRolloutLine[]`, whose\n * reward and authenticity fields have already been validated. `search` is the\n * default split; held-out pairing requires an explicit opt-in, while `dev` and\n * `canary` remain evaluation-only.\n */\n\nimport type { MintedRolloutLine, RolloutSplit } from '../rollout/schema'\nimport type { DpoLookups } from './exporters'\nimport {\n admitUngatedByInvocation,\n type LineContextRequirement,\n type RolloutLineContext,\n trainableLineReward,\n} from './rollout-input'\n\nexport type PreferenceStrategy =\n | 'paired-by-scenario-and-seed'\n | 'paired-by-scenario'\n | 'top-vs-bottom'\n\nexport interface PreferenceTriple {\n /** The scenario (input) the variants were run against. */\n scenarioId: string\n /** RunRecord ids on each side, for traceability. */\n chosenRunId: string\n rejectedRunId: string\n /** Variant ids — load-bearing for the RL update. */\n chosenVariantId: string\n rejectedVariantId: string\n /** The score gap between chosen and rejected. Larger = stronger signal. */\n marginScore: number\n /**\n * Optional `(chosen_score, rejected_score)` pair for soft-margin DPO\n * variants. Omitted for `top-vs-bottom` runs that don't carry meaningful\n * scalar gaps.\n */\n scores?: { chosen: number; rejected: number }\n /** Tie-breaker — when multiple seeds match this scenario, the one used. */\n seed?: number\n /**\n * Free-form metadata propagated from the rollout lines, such as original\n * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.\n */\n meta: {\n chosenPromptHash: string\n rejectedPromptHash: string\n chosenConfigHash: string\n rejectedConfigHash: string\n chosenModel: string\n rejectedModel: string\n }\n}\n\nexport interface ExtractPreferencesOptions {\n strategy?: PreferenceStrategy\n /**\n * Minimum score gap required to admit a pair. Pairs below this are\n * dropped — they're noise, not signal. Default 0.05 (5% of [0,1]).\n */\n minMargin?: number\n /**\n * Optional split filter. Without one, only search is included.\n * Holdout requires `allowHeldOutTrainingData: true`; dev and canary are\n * evaluation-only.\n */\n split?: RolloutSplit\n /** Named opt-in required before held-out lines may be paired. */\n allowHeldOutTrainingData?: boolean\n}\n\nexport interface PreferenceExtractionReport {\n pairs: PreferenceTriple[]\n /** Number of (scenario, seed) cells inspected. */\n cellsInspected: number\n /** Number of pairs filtered by `minMargin`. */\n pairsBelowMargin: number\n /** Number of cells with only one variant (no comparison possible). */\n cellsSingleton: number\n /** Strategy used. */\n strategy: PreferenceStrategy\n /**\n * Lines dropped before pairing because they carry no `candidate_id`. A\n * preference is a statement about two candidates, so a line that names none\n * cannot be paired.\n */\n linesWithoutCandidateId: number\n}\n\n/** The split each path pairs by default: training data comes from search. */\nconst SPLIT_DEFAULT: RolloutSplit = 'search'\n\n/**\n * The only shape the pairing strategies see.\n */\ninterface PairingCandidate {\n scenarioId: string\n runId: string\n candidateId: string\n /** null when a line records no seed. */\n seed: number | null\n score: number\n promptHash: string\n configHash: string\n model: string\n}\n\n/**\n * Convert rollout lines to preference triples for RL training.\n *\n * Returns a structured report so callers can see how much data was\n * dropped and why (low-margin pairs, singleton cells). For production\n * pipelines, you usually want to:\n *\n * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds\n * 2. Mint the runs with `mintRolloutRows` and call this with\n * `strategy: 'paired-by-scenario-and-seed'`\n * 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with\n * prompt/completion resolvers and pipe to your DPO trainer\n *\n * The gate is what makes a preference dataset safe: ordered on an ungated\n * score, a gamed run with an inflated number becomes the `chosen` side and DPO\n * is trained to prefer the gaming trajectory over its honest sibling. A gated\n * line arrives here already scored 0, so it sinks to `rejected`.\n */\nexport function extractPreferences(\n lines: MintedRolloutLine[],\n opts: ExtractPreferencesOptions = {},\n): PreferenceExtractionReport {\n const strategy = opts.strategy ?? 'paired-by-scenario-and-seed'\n const minMargin = opts.minMargin ?? 0.05\n const requestedSplit = opts.split\n if (requestedSplit === 'holdout' && opts.allowHeldOutTrainingData !== true) {\n throw new Error('extractPreferences: split \"holdout\" requires allowHeldOutTrainingData: true')\n }\n if (requestedSplit === 'dev' || requestedSplit === 'canary') {\n throw new Error(\n `extractPreferences: split \"${requestedSplit}\" is evaluation-only; train from \"search\"`,\n )\n }\n const candidates = candidatesFromLines(lines, opts)\n const report = pairCandidates(candidates.rows, strategy, minMargin)\n return { ...report, linesWithoutCandidateId: candidates.withoutCandidateId }\n}\n\ninterface NormalizedInput {\n rows: PairingCandidate[]\n withoutCandidateId: number\n}\n\nfunction candidatesFromLines(\n lines: MintedRolloutLine[],\n opts: ExtractPreferencesOptions,\n): NormalizedInput {\n const split = opts.split ?? SPLIT_DEFAULT\n const rows: PairingCandidate[] = []\n let withoutCandidateId = 0\n for (const line of lines) {\n if (line.task.split !== split) continue\n if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {\n continue\n }\n const score = trainableLineReward(line)\n if (score === null) continue\n const candidateId = line.candidate_id\n if (candidateId === null || candidateId === undefined || candidateId.length === 0) {\n withoutCandidateId++\n continue\n }\n rows.push({\n scenarioId: line.task.instance_id,\n runId: line.run_id,\n candidateId,\n seed: line.task.seed,\n score,\n // `policy.*` is nullable on the wire; a minted line always carries these\n // (RunRecord makes them mandatory). Empty string marks \"not recorded\" so\n // `toTRLFormat`'s hash lookup fails visibly instead of silently matching.\n promptHash: line.policy.prompt_hash ?? '',\n configHash: line.policy.config_hash ?? '',\n model: line.policy.model ?? '',\n })\n }\n return { rows, withoutCandidateId }\n}\n\nfunction pairCandidates(\n scoredEntries: PairingCandidate[],\n strategy: PreferenceStrategy,\n minMargin: number,\n): Omit<PreferenceExtractionReport, 'linesWithoutCandidateId'> {\n const pairs: PreferenceTriple[] = []\n let pairsBelowMargin = 0\n let cellsSingleton = 0\n let cellsInspected = 0\n\n if (strategy === 'paired-by-scenario-and-seed') {\n // Group by the canonical (scenarioId, seed) identity.\n const groups = new Map<string, PairingCandidate[]>()\n for (const e of scoredEntries) {\n const key = `${e.scenarioId}::${e.seed}`\n const arr = groups.get(key) ?? []\n arr.push(e)\n groups.set(key, arr)\n }\n\n for (const members of groups.values()) {\n cellsInspected++\n if (members.length < 2) {\n cellsSingleton++\n continue\n }\n for (let i = 0; i < members.length; i++) {\n for (let j = i + 1; j < members.length; j++) {\n const a = members[i]!\n const b = members[j]!\n if (a.candidateId === b.candidateId) continue\n const result = makePair(a, b, a.scenarioId, minMargin)\n if (result.kind === 'admit') pairs.push(result.pair)\n else pairsBelowMargin++\n }\n }\n }\n } else if (strategy === 'paired-by-scenario') {\n // Group by scenarioId → average per (variantId, scenarioId) across seeds.\n const byScenarioVariant = new Map<\n string,\n Map<string, { entry: PairingCandidate; sum: number; n: number }>\n >()\n for (const e of scoredEntries) {\n let perScenario = byScenarioVariant.get(e.scenarioId)\n if (!perScenario) {\n perScenario = new Map()\n byScenarioVariant.set(e.scenarioId, perScenario)\n }\n const cur = perScenario.get(e.candidateId)\n if (cur) {\n cur.sum += e.score\n cur.n++\n } else perScenario.set(e.candidateId, { entry: e, sum: e.score, n: 1 })\n }\n for (const [sid, perVariant] of byScenarioVariant.entries()) {\n cellsInspected++\n const arr = [...perVariant.values()].map((agg) => ({\n ...agg.entry,\n score: agg.sum / agg.n,\n }))\n if (arr.length < 2) {\n cellsSingleton++\n continue\n }\n for (let i = 0; i < arr.length; i++) {\n for (let j = i + 1; j < arr.length; j++) {\n const result = makePair(arr[i]!, arr[j]!, sid, minMargin)\n if (result.kind === 'admit') pairs.push(result.pair)\n else pairsBelowMargin++\n }\n }\n }\n } else {\n // top-vs-bottom: per scenario, top vs bottom only.\n const byScenario = new Map<string, PairingCandidate[]>()\n for (const e of scoredEntries) {\n const arr = byScenario.get(e.scenarioId) ?? []\n arr.push(e)\n byScenario.set(e.scenarioId, arr)\n }\n for (const [sid, arr] of byScenario.entries()) {\n cellsInspected++\n if (arr.length < 2) {\n cellsSingleton++\n continue\n }\n const sorted = [...arr].sort((a, b) => a.score - b.score)\n const top = sorted[sorted.length - 1]!\n const bot = sorted[0]!\n if (top.candidateId === bot.candidateId) {\n cellsSingleton++\n continue\n }\n const result = makePair(bot, top, sid, minMargin)\n if (result.kind === 'admit') pairs.push(result.pair)\n else pairsBelowMargin++\n }\n }\n\n return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy }\n}\n\nconst PREFERENCE_RUN_IDS = (t: PreferenceTriple): readonly string[] => [\n t.chosenRunId,\n t.rejectedRunId,\n]\n\nconst TRL_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'TRL preference export',\n contextType: 'RolloutLineContext',\n because:\n 'a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair.',\n}\n\nconst ANTHROPIC_CONTEXT_REQUIREMENT: LineContextRequirement = {\n exporter: 'Anthropic preference export',\n contextType: 'RolloutLineContext',\n because:\n 'a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one.',\n}\n\n/**\n * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`\n * where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes\n * would optimize the policy toward emitting hex digests. Neither the prompt\n * nor the completions live on the triple (it carries only run ids and hashes),\n * so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`\n * takes, keyed by run id, and this function resolves real text.\n *\n * The chosen and rejected sides of a valid pair share one prompt; resolving\n * both and comparing catches lookup bugs (a stale map keyed by the wrong id)\n * before they ship a row whose prompt does not match its rejected completion.\n *\n * `context` is REQUIRED: this is the third exporter over the identical\n * line-less input class, and the round that hardened `toPrmRows` while leaving\n * `toDpoRows` open is why every one of them now takes the same argument and\n * runs the same admission rule.\n */\nexport async function toTRLFormat(\n triples: PreferenceTriple[],\n lookups: DpoLookups,\n context: RolloutLineContext,\n): Promise<Array<{ prompt: string; chosen: string; rejected: string }>> {\n const admitted = admitUngatedByInvocation(\n triples,\n PREFERENCE_RUN_IDS,\n context,\n TRL_CONTEXT_REQUIREMENT,\n )\n const out: Array<{ prompt: string; chosen: string; rejected: string }> = []\n for (const t of admitted) {\n const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([\n Promise.resolve(lookups.promptOf(t.chosenRunId)),\n Promise.resolve(lookups.promptOf(t.rejectedRunId)),\n Promise.resolve(lookups.completionOf(t.chosenRunId)),\n Promise.resolve(lookups.completionOf(t.rejectedRunId)),\n ])\n if (chosenPrompt !== rejectedPrompt) {\n throw new Error(\n `toTRLFormat: preference \"${t.chosenRunId}\"/\"${t.rejectedRunId}\" resolves to different prompts`,\n )\n }\n out.push({ prompt: chosenPrompt, chosen, rejected })\n }\n return out\n}\n\n/**\n * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`\n * shape. Same caveat as TRL: prompt + outputs are content the caller has\n * to map back from the run record / raw event log.\n *\n * `context` is REQUIRED — see `toTRLFormat`. The emitted `margin` is a number\n * derived from the two runs' rewards, so this row is training signal even\n * though it ships no completion text.\n */\nexport function toAnthropicFormat(\n triples: PreferenceTriple[],\n context: RolloutLineContext,\n): Array<{ scenarioId: string; chosenRunId: string; rejectedRunId: string; margin: number }> {\n return admitUngatedByInvocation(\n triples,\n PREFERENCE_RUN_IDS,\n context,\n ANTHROPIC_CONTEXT_REQUIREMENT,\n ).map((t) => ({\n scenarioId: t.scenarioId,\n chosenRunId: t.chosenRunId,\n rejectedRunId: t.rejectedRunId,\n margin: t.marginScore,\n }))\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction makePair(\n a: PairingCandidate,\n b: PairingCandidate,\n scenarioId: string,\n minMargin: number,\n): { kind: 'admit'; pair: PreferenceTriple } | { kind: 'reject' } {\n const margin = Math.abs(a.score - b.score)\n if (margin < minMargin) return { kind: 'reject' }\n const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a]\n const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : undefined\n return {\n kind: 'admit',\n pair: {\n scenarioId,\n chosenRunId: chosen.runId,\n rejectedRunId: rejected.runId,\n chosenVariantId: chosen.candidateId,\n rejectedVariantId: rejected.candidateId,\n marginScore: chosen.score - rejected.score,\n scores: { chosen: chosen.score, rejected: rejected.score },\n seed,\n meta: {\n chosenPromptHash: chosen.promptHash,\n rejectedPromptHash: rejected.promptHash,\n chosenConfigHash: chosen.configHash,\n rejectedConfigHash: rejected.configHash,\n chosenModel: chosen.model,\n rejectedModel: rejected.model,\n },\n },\n }\n}\n","/**\n * Process reward extraction — step-level credit assignment from trace spans.\n *\n * RL on long-horizon agents needs *step-level* rewards, not run-level\n * ones. The classic credit-assignment problem (Sutton & Barto) requires\n * knowing which sub-decisions in a trajectory contributed to the\n * outcome. Modern systems (DeepSeek-R1, OpenAI o-series, Lightman et al.\n * \"Let's Verify Step by Step\" 2023) train *process reward models* (PRMs)\n * that score every step, then do RL with the PRM as the reward signal.\n *\n * This module extracts `StepReward[]` from trace spans — one per\n * meaningful step — and ships:\n *\n * 1. `extractStepRewards(store, runId, opts)` — span → step-reward\n * conversion using configurable per-span scorers (LLM judge over the\n * span output, deterministic checkers, or a learned PRM).\n * 2. `runwiseStepRewardSummary(stepRewards)` — aggregate the per-step\n * signal into a credit-assignment-aware run-level score.\n * 3. `prmTrainingPairs(stepRewards, options)` — produce the\n * `(prefix, suffix_chosen, suffix_rejected)` triples that PRM\n * training pipelines consume.\n *\n * What we ship: the *extraction* and *aggregation* infrastructure plus\n * the data shape PRM training expects. We do NOT ship the actual PRM\n * training (gradient descent over a transformer is out of scope for a\n * TS package). The interface is the contract; downstream consumers wire\n * their preferred trainer.\n *\n * Caveat the panel will land: this is descriptive credit assignment\n * (which steps correlate with outcome), not causal credit assignment\n * (which steps caused outcome). For causal claims you need\n * counterfactual rollouts or a learned dynamics model. Future work; the\n * descriptive version is what production PRM training actually uses.\n */\n\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface StepReward {\n /** Trace span this reward attaches to. */\n spanId: string\n runId: string\n /** Index in the trajectory (0-based, in started-at order). */\n stepIndex: number\n /** Span kind (typically 'tool', 'llm', 'judge'). */\n kind: Span['kind']\n /** Span name — for the consumer's downstream filtering. */\n name: string\n /** Step-level reward in [0, 1]. */\n reward: number\n /**\n * Determinism class. Mirrors the verifiable-reward distinction:\n * deterministic = test/compile/schema check; probabilistic = LLM judge.\n */\n determinism: 'deterministic' | 'probabilistic'\n /** Optional rationale / evidence — the trainer typically discards. */\n rationale?: string\n /** Optional weight — how much this step contributes to credit assignment. */\n weight?: number\n}\n\nexport interface StepScorer {\n /** Span kinds this scorer applies to. */\n appliesTo: Span['kind'][]\n /** Returns null to skip the span; returns a `StepReward` shape (without index/runId/spanId, which are filled in). */\n score(span: Span): Promise<Omit<StepReward, 'spanId' | 'runId' | 'stepIndex'>> | null | undefined\n}\n\nexport interface ExtractStepRewardsOptions {\n /**\n * Ordered list of scorers. Each span runs through scorers in order;\n * the first non-null result wins. If no scorer applies, the span is\n * skipped (not all spans are training-worthy).\n */\n scorers: StepScorer[]\n /** Optional filter — return null to drop the span entirely before scoring. */\n preFilter?: (span: Span) => boolean\n}\n\nexport async function extractStepRewards(\n store: TraceStore,\n runId: string,\n opts: ExtractStepRewardsOptions,\n): Promise<StepReward[]> {\n const spans = await store.spans({ runId })\n const ordered = [...spans].sort((a, b) => a.startedAt - b.startedAt)\n const out: StepReward[] = []\n let idx = 0\n for (const span of ordered) {\n if (opts.preFilter && !opts.preFilter(span)) continue\n let scored: Awaited<ReturnType<StepScorer['score']>> = null\n for (const s of opts.scorers) {\n if (!s.appliesTo.includes(span.kind)) continue\n const r = await s.score(span)\n if (r) {\n scored = r\n break\n }\n }\n if (!scored) continue\n out.push({\n spanId: span.spanId,\n runId,\n stepIndex: idx++,\n kind: span.kind,\n name: span.name,\n reward: scored.reward,\n determinism: scored.determinism,\n rationale: scored.rationale,\n weight: scored.weight,\n })\n }\n return out\n}\n\nexport interface RunwiseStepSummary {\n runId: string\n totalSteps: number\n meanReward: number\n /** Sum-of-rewards (weighted by `weight ?? 1`). Use as the run-level proxy. */\n sumWeightedReward: number\n /** Fraction of steps where reward < 0.5 — proxy for \"where the policy was wrong.\" */\n failureFraction: number\n /** Maximum drop in reward between consecutive steps — diagnoses a step where things went sideways. */\n worstStepDelta: number\n worstStepIndex: number | null\n}\n\nexport function runwiseStepRewardSummary(stepRewards: StepReward[]): RunwiseStepSummary {\n if (stepRewards.length === 0) {\n return {\n runId: '',\n totalSteps: 0,\n meanReward: 0,\n sumWeightedReward: 0,\n failureFraction: 0,\n worstStepDelta: 0,\n worstStepIndex: null,\n }\n }\n const runId = stepRewards[0]!.runId\n let sumW = 0\n let sumWR = 0\n let failures = 0\n let worstDelta = 0\n let worstIdx: number | null = null\n let prev = stepRewards[0]!.reward\n for (let i = 0; i < stepRewards.length; i++) {\n const s = stepRewards[i]!\n const w = s.weight ?? 1\n sumW += w\n sumWR += w * s.reward\n if (s.reward < 0.5) failures++\n if (i > 0) {\n const delta = s.reward - prev\n if (delta < worstDelta) {\n worstDelta = delta\n worstIdx = i\n }\n prev = s.reward\n } else {\n prev = s.reward\n }\n }\n return {\n runId,\n totalSteps: stepRewards.length,\n meanReward: sumW === 0 ? 0 : sumWR / sumW,\n sumWeightedReward: sumWR,\n failureFraction: failures / stepRewards.length,\n worstStepDelta: worstDelta,\n worstStepIndex: worstIdx,\n }\n}\n\nexport interface PrmTrainingTriple {\n /** Prefix run-id (or composite key) — the trajectory up to step k-1. */\n prefixRunId: string\n prefixStepIndex: number\n /** The step that came next on a high-reward trajectory. */\n chosenSpanId: string\n chosenReward: number\n /** A step from a divergent low-reward trajectory at the same prefix length. */\n rejectedSpanId: string\n rejectedReward: number\n /** The prefix run came from this run; the rejected step came from `rejectedRunId`. */\n rejectedRunId: string\n marginScore: number\n}\n\n/**\n * Build PRM training triples. The shape: pair runs that share an early\n * prefix (same scenario, same first N steps) and diverge later — at the\n * point of divergence, the high-reward run's next step is `chosen`, the\n * low-reward run's next step is `rejected`. This is the canonical PRM\n * training data shape from Lightman et al. and DeepSeek-R1 process\n * supervision.\n *\n * Implementation note: we don't have a way to detect \"same prefix\" in\n * the general agent setting (token-level prefixes require hashing model\n * outputs). The current heuristic groups by `(scenarioId, prefixSpanName\n * sequence)` — runs are paired when their first K span names match. For\n * production use this should be replaced with a proper trajectory-prefix\n * hash; the heuristic is good enough for early-stage scaffolding.\n */\nexport function prmTrainingPairs(\n stepRewardsByRun: Map<string, StepReward[]>,\n opts: { minMargin?: number; minPrefixLength?: number } = {},\n): PrmTrainingTriple[] {\n const minMargin = opts.minMargin ?? 0.2\n const minPrefix = opts.minPrefixLength ?? 1\n const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({ runId, steps }))\n const triples: PrmTrainingTriple[] = []\n\n for (let i = 0; i < runs.length; i++) {\n for (let j = i + 1; j < runs.length; j++) {\n const a = runs[i]!\n const b = runs[j]!\n const minLen = Math.min(a.steps.length, b.steps.length)\n if (minLen < minPrefix + 1) continue\n\n // Find the first index where the trajectories diverge: either by\n // step structure (kind/name mismatch) OR by reward gap ≥ minMargin.\n // Names that match but rewards that differ ARE divergence — that's\n // the canonical PRM training case (same step structure, different\n // outcomes via state/context).\n let divergenceIdx = -1\n for (let k = 0; k < minLen; k++) {\n const sa = a.steps[k]!\n const sb = b.steps[k]!\n const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name\n const rewardGap = Math.abs(sa.reward - sb.reward)\n if (structuralDivergence || rewardGap >= minMargin) {\n divergenceIdx = k\n break\n }\n }\n if (divergenceIdx < 0) continue\n if (divergenceIdx < minPrefix) continue\n\n const aNext = a.steps[divergenceIdx]!\n const bNext = b.steps[divergenceIdx]!\n const margin = Math.abs(aNext.reward - bNext.reward)\n if (margin < minMargin) continue\n\n const chosen = aNext.reward > bNext.reward ? aNext : bNext\n const rejected = aNext.reward > bNext.reward ? bNext : aNext\n const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId\n const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId\n triples.push({\n prefixRunId: chosenRun,\n prefixStepIndex: divergenceIdx - 1,\n chosenSpanId: chosen.spanId,\n chosenReward: chosen.reward,\n rejectedSpanId: rejected.spanId,\n rejectedReward: rejected.reward,\n rejectedRunId: rejectedRun,\n marginScore: chosen.reward - rejected.reward,\n })\n }\n }\n return triples\n}\n","/**\n * `runRLCampaign` — top-level orchestrator that runs the matrix and\n * produces every RL-ready artifact in one call.\n *\n * Wires:\n * 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)\n * 2. `extractVerifiableRewardsFromRecords` over the runs, separating deterministic\n * from probabilistic reward sources for the trainer\n * 3. `extractPreferences` to produce DPO/PPO/KTO triples\n * 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)\n * 5. `rubricPredictiveValidity` against an outcome store, when provided\n * 6. `detectRewardHacking` as a standing hygiene check\n * 7. Trainer-format export rows ready for prime-rl / TRL / verl\n *\n * The output `RLCampaignResult` is a single, audit-ready artifact: every\n * stage's output is in there. The consumer's downstream fits in a single\n * line: pass `result.preferences.pairs` to a DPO trainer,\n * `result.trainerRows.grpo` to GRPO, or `result.campaign.runs` plus\n * `result.rewardSignals` to a custom RL loop.\n */\n\nimport {\n type EvalCampaignOptions,\n type EvalCampaignResult,\n type FailedRun,\n runEvalCampaign,\n} from '../eval-campaign'\nimport type { OutcomeStore } from '../meta-eval/outcome-store'\nimport {\n type RubricPredictiveValidityReport,\n rubricPredictiveValidity,\n} from '../meta-eval/rubric-predictive-validity'\nimport { mintRolloutRows } from '../rollout/mint'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport { evaluateInterimReleaseConfidence, type InterimReleaseConfidence } from '../sequential'\nimport { InMemoryTraceStore } from '../trace/store'\nimport {\n type DpoExportRow,\n type DpoLookups,\n type GrpoExportRow,\n type GrpoLookups,\n type SftExportRow,\n type SftLookups,\n toDpoRows,\n toGrpoRows,\n toSftRows,\n} from './exporters'\nimport {\n type ExtractPreferencesOptions,\n extractPreferences,\n type PreferenceExtractionReport,\n} from './preferences'\nimport { detectRewardHacking, type RewardHackingReport } from './reward-hacking'\nimport {\n extractVerifiableRewardsFromRecords,\n type VerifiableReward,\n type VerifiableRewardExtractionOptions,\n} from './verifiable-reward'\n\nexport interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {\n /** Preference-extraction options. Default uses paired-by-scenario-and-seed with min-margin 0.05. */\n preferences?: ExtractPreferencesOptions\n /** Verifiable-reward extraction options. */\n verifiableReward?: VerifiableRewardExtractionOptions\n /** Outcome store + metric names — when supplied, runs `rubricPredictiveValidity` post-campaign. */\n outcomeStore?: OutcomeStore\n outcomeMetrics?: string[]\n /** Anytime-valid sequential evaluation options. */\n sequential?: {\n alpha?: number\n bound?: number\n rope?: { low: number; high: number }\n /**\n * Smallest acceptable `answered / dealt` fraction of paired cells, required\n * of EVERY candidate before the interim verdict is computed at all. Default\n * 1: every cell the comparison was dealt must carry a score on both arms.\n *\n * The default is 1 because `recommendation.decision` can be `promote_now`,\n * and a recommendation computed over \"the cells that happened to pair\" is\n * computed over a set the candidate selected by failing. Below this,\n * `interimConfidence` is null and `deltaCoverage` says by how much — never a\n * silent 0. Must be in [0, 1]; anything else throws rather than clamping.\n */\n minDeltaCoverage?: number\n }\n /** Trainer-format export lookups. When provided, the orchestrator builds the corresponding rows. */\n trainerExport?: {\n dpo?: DpoLookups\n grpo?: GrpoLookups\n sft?: SftLookups\n }\n}\n\n/**\n * How much of one candidate's DEALT paired work produced a usable delta.\n *\n * `answered + unscoredCandidate + unscoredComparator + unmatched === dealt` — a\n * complete, mutually exclusive partition of the (scenarioId, seed) cells the\n * comparison was dealt, so no cell can leave the delta series without appearing\n * in exactly one bucket.\n */\nexport interface PairedDeltaCoverage {\n candidateId: string\n /** Cells the comparison was dealt: a run on either arm. */\n dealt: number\n /** Dealt cells scored on BOTH arms — the deltas the verdict is read from. */\n answered: number\n /** Dealt cells this candidate ran and produced no usable score for. */\n unscoredCandidate: number\n /** Dealt cells the comparator ran and produced no usable score for. */\n unscoredComparator: number\n /** Dealt cells only one of the two arms produced a run for. */\n unmatched: number\n /** `answered / dealt`, or 0 when nothing was dealt. */\n coverage: number\n}\n\nexport interface RLCampaignResult {\n campaign: EvalCampaignResult\n /** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */\n rewardSignals: Array<{ runId: string; reward: VerifiableReward | null }>\n /** Preference extraction report. */\n preferences: PreferenceExtractionReport\n /** Anytime-valid interim verdict over the paired deltas (vs comparator).\n * Null when no comparator was configured, when nothing paired, or when a\n * candidate fell below `sequential.minDeltaCoverage` — read `deltaCoverage`\n * to tell those apart. */\n interimConfidence: InterimReleaseConfidence | null\n /** Answered / dealt paired cells per candidate — the denominator behind\n * `interimConfidence`, reported on EVERY path including the ones where the\n * verdict was refused. Empty when no comparator was configured. */\n deltaCoverage: PairedDeltaCoverage[]\n /** Standing reward-hacking hygiene check. */\n rewardHacking: RewardHackingReport\n /** Predictive validity, when an outcome store was supplied. */\n predictiveValidity: RubricPredictiveValidityReport | null\n /** Trainer-export rows, populated only for the formats the caller requested via `trainerExport`. */\n trainerRows: {\n dpo?: DpoExportRow[]\n grpo?: GrpoExportRow[]\n sft?: SftExportRow[]\n }\n /**\n * One-line top-level summary the consumer can log.\n */\n summary: string\n /**\n * Convenience type-tag — consumers can branch on `result.kind`.\n */\n kind: 'agent-eval-rl-campaign'\n}\n\nexport async function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult> {\n const splitTag = opts.splitTag ?? 'search'\n\n // ── 1. Run the matrix ──────────────────────────────────────────────\n const campaign = await runEvalCampaign({ ...opts, splitTag })\n\n // ── 2. Extract reward signals (deterministic-first) ────────────────\n const rewardSignals = extractVerifiableRewardsFromRecords(\n campaign.runs,\n opts.verifiableReward ?? {},\n )\n\n // ── 3. Mint the scored runs once, then derive all training artifacts ──\n const scoredRuns = campaign.runs.filter((run) => runTaskScore(run) !== undefined)\n const { rows: rolloutLines } = await mintRolloutRows(scoredRuns, new InMemoryTraceStore())\n const preferences = extractPreferences(rolloutLines, {\n ...opts.preferences,\n strategy: opts.preferences?.strategy ?? 'paired-by-scenario-and-seed',\n minMargin: opts.preferences?.minMargin ?? 0.05,\n split: opts.preferences?.split ?? splitTag,\n })\n\n // ── 4. Sequential / anytime-valid interim verdict ──────────────────\n let interimConfidence: InterimReleaseConfidence | null = null\n let deltaCoverage: PairedDeltaCoverage[] = []\n const minDeltaCoverage = opts.sequential?.minDeltaCoverage ?? 1\n if (!(Number.isFinite(minDeltaCoverage) && minDeltaCoverage >= 0 && minDeltaCoverage <= 1)) {\n throw new Error(\n `runRLCampaign: sequential.minDeltaCoverage must be a finite fraction in [0, 1], got ${minDeltaCoverage}`,\n )\n }\n if (opts.report?.comparator) {\n const comparator = opts.report.comparator\n const series = collectPairedDeltaSeries(campaign.runs, campaign.failedRuns, comparator)\n deltaCoverage = series.map((s) => s.coverage)\n // Fail closed on a shrunken denominator: the recommendation can be\n // `promote_now`, so it does not get computed over the cells that happened\n // to pair. The accounting ships either way.\n const covered = series.every((s) => s.coverage.coverage >= minDeltaCoverage)\n if (covered && series.some((s) => s.deltas.length > 0)) {\n interimConfidence = evaluateInterimReleaseConfidence({\n deltaSeries: series.map(({ candidateId, deltas }) => ({ candidateId, deltas })),\n alpha: opts.sequential?.alpha,\n bound: opts.sequential?.bound,\n rope: opts.sequential?.rope ?? opts.report?.rope,\n })\n }\n }\n\n // ── 5. Standing reward-hacking hygiene ─────────────────────────────\n const rewardHacking = detectRewardHacking({\n runs: campaign.runs,\n verifiableRewardOptions: opts.verifiableReward,\n })\n\n // ── 6. Predictive validity (when outcomes are supplied) ────────────\n let predictiveValidity: RubricPredictiveValidityReport | null = null\n if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) {\n predictiveValidity = await rubricPredictiveValidity({\n runs: campaign.runs,\n outcomes: opts.outcomeStore,\n outcomeMetrics: opts.outcomeMetrics,\n })\n }\n\n // ── 7. Trainer-format export ───────────────────────────────────────\n const trainerRows: RLCampaignResult['trainerRows'] = {}\n if (opts.trainerExport?.dpo) {\n trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, {\n lines: rolloutLines,\n })\n }\n if (opts.trainerExport?.grpo) {\n trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo)\n }\n if (opts.trainerExport?.sft) {\n trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft)\n }\n\n const summary = buildSummary({\n campaign,\n preferences,\n interimConfidence,\n deltaCoverage,\n rewardHacking,\n predictiveValidity,\n })\n\n return {\n campaign,\n rewardSignals,\n preferences,\n interimConfidence,\n deltaCoverage,\n rewardHacking,\n predictiveValidity,\n trainerRows,\n summary,\n kind: 'agent-eval-rl-campaign',\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\n/**\n * Pair on (scenarioId, seed) and count what was DEALT, not only what paired.\n *\n * Every drop below used to be silent: a comparator run with no score never\n * entered the map, a candidate run with no score was skipped, and a candidate\n * cell with no comparator at the same identity was skipped. The surviving\n * deltas then flowed into `evaluateInterimReleaseConfidence`, whose\n * `recommendation.decision` can be `promote_now` — so a candidate that scored 6\n * of 26 cells could be recommended for promotion off a series of length 6, with\n * nothing in the result saying 20 cells went dark. Same defect as the promotion\n * gates, one call frame up.\n *\n * The denominator is MEASURED: a cell counts as dealt because a run for it\n * exists on either arm. Missing scores are not imputed — the caller who knows\n * the failure value of its metric writes it onto the record.\n */\nfunction collectPairedDeltaSeries(\n runs: RunRecord[],\n failedRuns: FailedRun[],\n comparator: string,\n): Array<{ candidateId: string; deltas: number[]; coverage: PairedDeltaCoverage }> {\n const cellKey = (r: { scenarioId: string; seed: number }) => `${r.scenarioId}::${r.seed}`\n // A cell that failed integrity or crashed never reaches `campaign.runs` at\n // all, so counting only the surviving records would make the very failure this\n // check exists to catch invisible. `failedRuns` carries the same\n // (variantId, scenarioId, seed) identity — it is dealt work that produced no\n // score, which is exactly what the denominator must hold.\n const dealtFromFailures = new Map<string, Set<string>>()\n for (const f of failedRuns) {\n let set = dealtFromFailures.get(f.variantId)\n if (!set) {\n set = new Set<string>()\n dealtFromFailures.set(f.variantId, set)\n }\n set.add(cellKey(f))\n }\n // Comparator side, split into what it was dealt and what it answered.\n const comparatorDealt = new Set<string>(dealtFromFailures.get(comparator) ?? [])\n const comparatorScore = new Map<string, number>()\n for (const r of runs) {\n if (r.candidateId !== comparator) continue\n const key = cellKey(r)\n comparatorDealt.add(key)\n // Ungated (`runTaskScore` is raw): this is a measurement of the paired\n // delta between candidates, not a value any trainer consumes. Gating it\n // would report a candidate as worse than it measured; the gamed run should\n // be excluded upstream instead.\n const score = runTaskScore(r)\n if (score === undefined) continue\n comparatorScore.set(key, score)\n }\n const dealtByCandidate = new Map<string, Set<string>>()\n const scoreByCandidate = new Map<string, Map<string, number>>()\n for (const [variantId, cells] of dealtFromFailures) {\n if (variantId === comparator) continue\n dealtByCandidate.set(variantId, new Set(cells))\n }\n for (const r of runs) {\n if (r.candidateId === comparator) continue\n const key = cellKey(r)\n let dealt = dealtByCandidate.get(r.candidateId)\n if (!dealt) {\n dealt = new Set<string>()\n dealtByCandidate.set(r.candidateId, dealt)\n }\n dealt.add(key)\n const score = runTaskScore(r)\n if (score === undefined) continue\n let scored = scoreByCandidate.get(r.candidateId)\n if (!scored) {\n scored = new Map<string, number>()\n scoreByCandidate.set(r.candidateId, scored)\n }\n scored.set(key, score)\n }\n\n return [...dealtByCandidate.entries()].map(([candidateId, candidateDealt]) => {\n const scored = scoreByCandidate.get(candidateId) ?? new Map<string, number>()\n const deltas: number[] = []\n let unscoredCandidate = 0\n let unscoredComparator = 0\n let unmatched = 0\n // The dealt set is the union: a cell the comparator ran and this candidate\n // never wrote a row for is still work the comparison was given.\n for (const key of new Set([...candidateDealt, ...comparatorDealt])) {\n const onCandidate = candidateDealt.has(key)\n const onComparator = comparatorDealt.has(key)\n if (!onCandidate || !onComparator) {\n unmatched += 1\n continue\n }\n const a = scored.get(key)\n const b = comparatorScore.get(key)\n if (a === undefined) unscoredCandidate += 1\n else if (b === undefined) unscoredComparator += 1\n else deltas.push(a - b)\n }\n const dealt = new Set([...candidateDealt, ...comparatorDealt]).size\n return {\n candidateId,\n deltas,\n coverage: {\n candidateId,\n dealt,\n answered: deltas.length,\n unscoredCandidate,\n unscoredComparator,\n unmatched,\n coverage: dealt === 0 ? 0 : deltas.length / dealt,\n },\n }\n })\n}\n\nfunction buildSummary(args: {\n campaign: EvalCampaignResult\n preferences: PreferenceExtractionReport\n interimConfidence: InterimReleaseConfidence | null\n deltaCoverage: PairedDeltaCoverage[]\n rewardHacking: RewardHackingReport\n predictiveValidity: RubricPredictiveValidityReport | null\n}): string {\n const c = args.campaign\n const lines = [\n `${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}…)`,\n `preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`,\n ]\n if (args.interimConfidence) {\n lines.push(\n `sequential verdict: ${args.interimConfidence.recommendation.decision}` +\n (args.interimConfidence.recommendation.candidateId\n ? ` ${args.interimConfidence.recommendation.candidateId}`\n : ''),\n )\n }\n // Never a silent 0 — a shrunken denominator has to say by how much, including\n // (especially) on the path where the verdict was refused for being shrunken.\n const shortfall = args.deltaCoverage.filter((c) => c.answered < c.dealt)\n if (shortfall.length > 0) {\n lines.push(\n `paired-delta coverage: ${shortfall\n .map((c) => `${c.candidateId} ${c.answered}/${c.dealt}`)\n .join(', ')}${args.interimConfidence ? '' : ' (sequential verdict withheld)'}`,\n )\n }\n lines.push(\n `reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`,\n )\n if (args.predictiveValidity) {\n const top = args.predictiveValidity.ranked[0]\n lines.push(\n `top-rubric: ${top?.rubric ?? 'none'} ρ=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? 'no data'})`,\n )\n }\n return lines.join(' | ')\n}\n\n// Re-export `runEvalCampaign` so consumers can pick the lower-level\n// primitive without flipping import paths.\nexport { runEvalCampaign } from '../eval-campaign'\n","/**\n * Adapters: convert measurement outputs into the canonical `RunRecord[]`\n * artifact that `replayCache`, `pairedEvalueSequence`, and\n * `rubricPredictiveValidity` consume. Two sources:\n * - `campaignToRunRecords` — the campaign substrate's per-cell results\n * (the modern path: `runCampaign` / `runImprovementLoop` → records).\n * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.\n *\n * Adapters are thin and explicit — every mandatory `RunRecord` field comes\n * from a caller-supplied context (`commitSha`, `model`, `promptHash`,\n * `configHash`) plus the cell's runtime data. The validator still rejects\n * bare-alias model strings — the caller snapshot-pins.\n */\n\nimport { campaignCellToRunRecord } from '../campaign/run-record'\nimport type { CampaignResult } from '../campaign/types'\nimport type { LayerResult, VerificationReport } from '../multi-layer-verifier'\nimport type { RunRecord, RunSplitTag } from '../run-record'\n\nexport interface AdapterContext {\n /** Logical experiment id — typically the campaign or sweep identifier. */\n experimentId: string\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** Hash of the effective prompt sent to the model. */\n promptHash: string\n /** Hash of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Default split tag. Default `'search'`. */\n splitTag?: RunSplitTag\n /** Estimated cost in USD when the source doesn't record one. */\n defaultCostUsd?: number\n}\n\n/**\n * Convert a `CampaignResult` into canonical `RunRecord[]`, one per cell.\n * Successful judged cells carry their mean judge composite and dimensions.\n * Errored or unjudged cells remain unlabeled while retaining explicit terminal\n * outcome, execution-error count, token usage, cost, and failure detail.\n * `candidateId` identifies the measured surface and defaults to the campaign\n * manifest hash.\n */\nexport function campaignToRunRecords(\n campaign: CampaignResult,\n ctx: AdapterContext & { candidateId?: string },\n): RunRecord[] {\n const splitTag = ctx.splitTag ?? 'search'\n const candidateId = ctx.candidateId ?? campaign.manifestHash\n return campaign.cells.map((cell) =>\n campaignCellToRunRecord(cell, {\n runId: cell.cellId,\n experimentId: ctx.experimentId,\n candidateId,\n model: ctx.model,\n promptHash: ctx.promptHash,\n configHash: ctx.configHash,\n commitSha: ctx.commitSha,\n splitTag,\n defaultCostUsd: ctx.defaultCostUsd,\n }),\n )\n}\n\n/**\n * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.\n * A split score is emitted only when `report.taskScore` proves the configured\n * scoring panel completed. Partial scores remain in `outcome.raw` for\n * diagnosis. Layer errors and timeouts become judge or execution telemetry;\n * only a scored `fail` layer may produce task-failure detail.\n */\nexport function verificationReportToRunRecord(\n report: VerificationReport,\n ctx: AdapterContext & { candidateId: string; scenarioId: string },\n opts: { runId?: string } = {},\n): RunRecord {\n const splitTag = ctx.splitTag ?? 'search'\n const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`\n const hasValidLayerMeasurement = report.layers.some(hasValidTaskMeasurement)\n const taskScore =\n hasValidLayerMeasurement && isValidScore(report.taskScore) ? report.taskScore : undefined\n let executionErrorCount = 0\n let judgeErrorCount = 0\n let layerErrorCount = 0\n let layerTimeoutCount = 0\n let unscoredLayerCount = 0\n\n const raw: Record<string, number> = {\n pass_count: report.passCount,\n fail_count: report.failCount,\n error_count: report.errorCount,\n skipped_count: report.skippedCount,\n duration_ms: report.durationMs,\n execution_error_count: 0,\n }\n for (const layer of report.layers) {\n if (hasValidTaskMeasurement(layer)) raw[`layer.${layer.layer}`] = layer.score\n else unscoredLayerCount++\n raw[`layer_${layer.layer}_pass`] = layer.status === 'pass' ? 1 : 0\n if (layer.status === 'error' || layer.status === 'timeout') {\n if (layer.errorSource === 'judge') judgeErrorCount++\n else executionErrorCount++\n if (layer.status === 'error') layerErrorCount++\n else layerTimeoutCount++\n }\n if (layer.diagnostics) {\n for (const [k, v] of Object.entries(layer.diagnostics)) {\n if (typeof v === 'number' && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v\n }\n }\n }\n\n raw.execution_error_count = executionErrorCount\n if (judgeErrorCount > 0) raw.judge_error_count = judgeErrorCount\n if (layerErrorCount > 0) raw.layer_error_count = layerErrorCount\n if (layerTimeoutCount > 0) raw.layer_timeout_count = layerTimeoutCount\n if (unscoredLayerCount > 0) raw.unscored_layer_count = unscoredLayerCount\n if (taskScore !== undefined) raw.blended_score = taskScore\n\n const firstScoredFailure = report.layers.find(\n (layer) => layer.status === 'fail' && hasValidTaskMeasurement(layer),\n )\n const outcome: RunRecord['outcome'] = { raw }\n if (taskScore !== undefined) {\n if (splitTag === 'holdout') outcome.holdoutScore = taskScore\n else outcome.searchScore = taskScore\n }\n\n return {\n runId,\n experimentId: ctx.experimentId,\n candidateId: ctx.candidateId,\n seed: 0,\n model: ctx.model,\n promptHash: ctx.promptHash,\n configHash: ctx.configHash,\n commitSha: ctx.commitSha,\n wallMs: report.durationMs,\n costUsd: ctx.defaultCostUsd ?? null,\n costProvenance:\n ctx.defaultCostUsd === undefined\n ? { kind: 'uncaptured', usd: null }\n : { kind: 'estimated', usd: ctx.defaultCostUsd },\n tokenUsage: { input: 0, output: 0 },\n terminalOutcome: 'succeeded',\n outcome,\n ...(firstScoredFailure\n ? {\n failureClass: 'unknown' as const,\n failureMode: `layer_${firstScoredFailure.layer}_fail`,\n }\n : {}),\n splitTag,\n scenarioId: ctx.scenarioId,\n }\n}\n\nfunction hasValidTaskMeasurement(\n layer: LayerResult,\n): layer is LayerResult & { status: 'pass' | 'fail'; score: number } {\n return (layer.status === 'pass' || layer.status === 'fail') && isValidScore(layer.score)\n}\n\nfunction isValidScore(score: unknown): score is number {\n return typeof score === 'number' && Number.isFinite(score) && score >= 0 && score <= 1\n}\n","/**\n * Simulator fidelity — score a user SIMULATOR's realism against real-user\n * trace distributions.\n *\n * Synthetic-persona evals (`PersonaConfig`-driven canonical evals, fuzz\n * user-simulator objectives) stand in for real users in most of the numbers\n * we publish. The standing threat is the Sim2Real gap: a simulator that is\n * distributionally unlike production creates \"easy mode\" and silently\n * inflates every score built on it. This module measures that gap from the\n * SAME artifact both sides already produce — `RunRecord`s — so no new\n * capture pipeline is needed:\n *\n * - `simFidelityReport` — per-feature Jensen-Shannon divergence between\n * simulated and production record distributions, collapsed into a\n * fidelity coefficient in [0,1].\n * - `easyModeCheck` — the headline academic failure mode (sim inflates\n * pass-rate over production) as its own named artifact.\n *\n * Every synthetic-persona eval result should publish its fidelity\n * coefficient alongside the score — a number from an unrepresentative\n * simulator is an unlabeled estimate. Wire-in points:\n *\n * - canonical persona evals: pass the campaign's `RunRecord`s as\n * `simulated` and intake-adapter output (`contract/intake`: OTel spans,\n * feedback tables, coding-agent sessions) as `production`\n * - the fuzz user-sim objective: use `1 - report.fidelity` as a realism\n * penalty when searching over generated personas\n * - the durable corpus (`./corpus`): both sides read straight from\n * `readCorpus` — tag sim vs production by `experimentId`\n */\n\nimport { ValidationError } from '../errors'\nimport { observedSplitScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport type { FailureClass } from '../trace/schema'\nimport type { CorpusRecord } from './corpus'\n\n/** Extracts a flat behavioral feature map from one record. `string` values\n * are categorical, `number` values are quantile-bucketed over the union of\n * both sides, `null` means the feature is absent on this record and is\n * counted explicitly as its own category (never silently dropped). */\nexport type BehaviorFeatures = (record: RunRecord) => Record<string, string | number | null>\n\n/** Reserved histogram category for `null` feature values. A capture-rate\n * difference (one side instruments a signal, the other does not) registers\n * as divergence by design: a simulator that produces no tool traces is not\n * representative of production that does. */\nexport const ABSENT_CATEGORY = '(absent)'\n\n/** Minimum non-null observations PER SIDE for a feature to enter the\n * fidelity mean. Below this the JSD estimate is sampling noise. */\nconst DEFAULT_MIN_N_PER_FEATURE = 20\n\n/** Quantile buckets used to discretize numeric features. Quartiles balance\n * resolution against per-bucket sample size at the default minN. */\nconst DEFAULT_QUANTILE_BUCKETS = 4\n\n/** Fidelity at or above this → 'representative'; below → 'skewed'.\n * 1 − 0.8 = mean JSD 0.2 ≈ distributions that mostly overlap with one\n * clearly shifted mode — the point where per-feature shifts start changing\n * which failure classes an eval can even observe. */\nconst REPRESENTATIVE_MIN_FIDELITY = 0.8\n\nconst TOP_SHIFT_COUNT = 5\n\n/**\n * Default feature set — ONLY fields verified present on both simulated and\n * production records:\n *\n * - `score`, `wall_ms`, `output_tokens` — mandatory per the `RunRecord`\n * validator (non-finite values read as absent rather than poisoning a\n * bucket).\n * - `failure_class` — optional taxonomy field; absent counted explicitly.\n * - `turn_count`, `tool_errors`, `tool_error_recovery` — derived from the\n * `outcome.raw` counters the intake adapters and eval harnesses write\n * (`turns_completed`, `assistant_messages`, `tool_errors`,\n * `turns_aborted`); absent on records whose producer did not capture\n * them, counted explicitly.\n * - `completion_length` — from the optional `CorpusRecord` trajectory\n * text; the message-length proxy when records come from the corpus.\n *\n * `RunRecord` carries event COUNTS, not event ordering, so\n * `tool_error_recovery` is a counts-only derivation: errors occurred and the\n * run still completed cleanly ('recovered') vs aborted or classified as a\n * failure ('unrecovered') — not a literal error→retry sequence check.\n */\nexport const defaultBehaviorFeatures: BehaviorFeatures = (record) => {\n const raw: Record<string, number> = record.outcome?.raw ?? {}\n const toolErrors = finiteOrNull(raw.tool_errors)\n const turnsAborted = finiteOrNull(raw.turns_aborted)\n const completion = (record as CorpusRecord).completion\n return {\n // RAW (`observedSplitScore`), deliberately: this feature vector is one\n // half of a sim-vs-production divergence measurement. Gating a gamed run to\n // 0 would move the simulated distribution toward production and report the\n // simulator as MORE faithful precisely where it is being gamed. Each split\n // is read separately rather than through `observedScore` so a non-finite\n // holdout score falls back to search instead of poisoning the bucket.\n score:\n finiteOrNull(observedSplitScore(record, 'holdout')) ??\n finiteOrNull(observedSplitScore(record, 'search')),\n failure_class: record.failureClass ?? null,\n wall_ms: finiteOrNull(record.wallMs),\n output_tokens: finiteOrNull(record.tokenUsage?.output),\n turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),\n tool_errors: toolErrors,\n tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),\n completion_length: typeof completion === 'string' ? completion.length : null,\n }\n}\n\nfunction toolErrorRecovery(\n toolErrors: number | null,\n turnsAborted: number | null,\n failureClass: FailureClass | undefined,\n): string | null {\n if (toolErrors === null) return null\n if (toolErrors === 0) return 'no-tool-errors'\n const failed =\n (turnsAborted ?? 0) > 0 || (failureClass !== undefined && failureClass !== 'success')\n return failed ? 'unrecovered' : 'recovered'\n}\n\nfunction finiteOrNull(value: unknown): number | null {\n return typeof value === 'number' && Number.isFinite(value) ? value : null\n}\n\n// ── Divergence core ──────────────────────────────────────────────────\n\n/**\n * Jensen-Shannon divergence between two categorical histograms (raw counts;\n * normalized internally). Log base 2 → bounded [0,1]: 0 = identical\n * distributions, 1 = disjoint support. Symmetric, defined even where the\n * supports differ — exactly the regime sim-vs-production comparison lives in.\n * Throws on zero-mass or negative/non-finite counts: an empty histogram has\n * no distribution and a silent 0 would read as \"perfectly representative\".\n */\nexport function jsDivergence(p: Record<string, number>, q: Record<string, number>): number {\n const keys = new Set([...Object.keys(p), ...Object.keys(q)])\n if (keys.size === 0) {\n throw new ValidationError('jsDivergence: both histograms are empty')\n }\n let pSum = 0\n let qSum = 0\n for (const key of keys) {\n const pv = p[key] ?? 0\n const qv = q[key] ?? 0\n if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) {\n throw new ValidationError(`jsDivergence: negative or non-finite count for category \"${key}\"`)\n }\n pSum += pv\n qSum += qv\n }\n if (pSum === 0 || qSum === 0) {\n throw new ValidationError('jsDivergence: a histogram with zero total mass has no distribution')\n }\n let divergence = 0\n for (const key of keys) {\n const pp = (p[key] ?? 0) / pSum\n const qp = (q[key] ?? 0) / qSum\n const m = (pp + qp) / 2\n if (pp > 0) divergence += 0.5 * pp * Math.log2(pp / m)\n if (qp > 0) divergence += 0.5 * qp * Math.log2(qp / m)\n }\n // float error can land epsilon outside [0,1]\n return Math.min(1, Math.max(0, divergence))\n}\n\n/**\n * Deterministic quantile edges over a value set (the UNION of both sides, so\n * sim and production land in the same buckets). Linear interpolation between\n * order statistics; duplicate edges from heavy ties collapse into fewer,\n * wider buckets. Returns `bucketCount - 1` edges before deduplication.\n */\nexport function quantileEdges(values: number[], bucketCount = DEFAULT_QUANTILE_BUCKETS): number[] {\n if (values.length === 0) {\n throw new ValidationError('quantileEdges: requires at least one value')\n }\n if (!Number.isInteger(bucketCount) || bucketCount < 2) {\n throw new ValidationError(\n `quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`,\n )\n }\n const sorted = [...values].sort((a, b) => a - b)\n const edges: number[] = []\n for (let k = 1; k < bucketCount; k++) {\n const pos = (k / bucketCount) * (sorted.length - 1)\n const lo = sorted[Math.floor(pos)]!\n const hi = sorted[Math.ceil(pos)]!\n edges.push(lo + (pos - Math.floor(pos)) * (hi - lo))\n }\n return [...new Set(edges)]\n}\n\n/** Stable half-open bucket label for a value against quantile edges:\n * `[-inf,e0)`, `[e0,e1)`, …, `[eLast,+inf)`. */\nexport function bucketLabel(value: number, edges: number[]): string {\n let i = 0\n while (i < edges.length && value >= edges[i]!) i++\n const lo = i === 0 ? '-inf' : String(edges[i - 1]!)\n const hi = i === edges.length ? '+inf' : String(edges[i]!)\n return `[${lo},${hi})`\n}\n\n// ── Fidelity report ──────────────────────────────────────────────────\n\nexport interface FeatureShift {\n /** Category label (a string value, a numeric bucket, or `ABSENT_CATEGORY`). */\n value: string\n /** Probability of this category among ALL simulated records (nulls included\n * via `ABSENT_CATEGORY`, so each side's shifts sum to 1). */\n pSim: number\n /** Probability among ALL production records. */\n pProd: number\n}\n\nexport interface FeatureDivergence {\n feature: string\n /** Jensen-Shannon divergence in [0,1] for this feature. */\n divergence: number\n /** Largest |pSim − pProd| categories, descending — where the sim deviates. */\n topShifts: FeatureShift[]\n /** Non-null observations on the simulated side. */\n nSim: number\n /** Non-null observations on the production side. */\n nProd: number\n}\n\nexport type FidelityVerdict = 'representative' | 'skewed' | 'insufficient-data'\n\nexport interface FidelityReport {\n perDimension: FeatureDivergence[]\n /** 1 − mean divergence over features with sufficient data. NaN when the\n * verdict is 'insufficient-data' — a 0 would read as \"maximally skewed\"\n * and silently poison downstream aggregation; check `verdict` first. */\n fidelity: number\n /** Features excluded because either side had fewer than `minNPerFeature`\n * non-null observations. Named, never silently dropped. */\n insufficientData: string[]\n /** 'representative' when fidelity >= REPRESENTATIVE_MIN_FIDELITY (0.8),\n * 'skewed' below, 'insufficient-data' when no feature met minN. */\n verdict: FidelityVerdict\n}\n\nexport interface SimFidelityOptions {\n /** Feature extractor. Defaults to `defaultBehaviorFeatures`. */\n features?: BehaviorFeatures\n /** Minimum non-null observations per side per feature. Default 20. */\n minNPerFeature?: number\n}\n\n/**\n * Compare a simulator's RunRecords against production RunRecords, feature by\n * feature. Numeric features are bucketed by deterministic quantiles of the\n * union; nulls count as an explicit `ABSENT_CATEGORY`. Throws on empty\n * inputs — \"no records\" is a wiring error, not a distribution.\n */\nexport function simFidelityReport(\n simulated: RunRecord[],\n production: RunRecord[],\n opts: SimFidelityOptions = {},\n): FidelityReport {\n if (simulated.length === 0) {\n throw new ValidationError('simFidelityReport: simulated records are empty')\n }\n if (production.length === 0) {\n throw new ValidationError('simFidelityReport: production records are empty')\n }\n const extract = opts.features ?? defaultBehaviorFeatures\n const minN = opts.minNPerFeature ?? DEFAULT_MIN_N_PER_FEATURE\n\n const simMaps = simulated.map(extract)\n const prodMaps = production.map(extract)\n\n // union of feature names in first-seen order — extractors may emit\n // different keys per record (e.g. domain-conditional features)\n const featureNames: string[] = []\n const seen = new Set<string>()\n for (const map of [...simMaps, ...prodMaps]) {\n for (const name of Object.keys(map)) {\n if (!seen.has(name)) {\n seen.add(name)\n featureNames.push(name)\n }\n }\n }\n\n const perDimension: FeatureDivergence[] = []\n const insufficientData: string[] = []\n\n for (const feature of featureNames) {\n const simVals = simMaps.map((m) => m[feature] ?? null)\n const prodVals = prodMaps.map((m) => m[feature] ?? null)\n const nSim = simVals.filter((v) => v !== null).length\n const nProd = prodVals.filter((v) => v !== null).length\n if (nSim < minN || nProd < minN) {\n insufficientData.push(feature)\n continue\n }\n const { sim, prod } = histograms(feature, simVals, prodVals)\n perDimension.push({\n feature,\n divergence: jsDivergence(sim, prod),\n topShifts: topShifts(sim, simVals.length, prod, prodVals.length),\n nSim,\n nProd,\n })\n }\n\n if (perDimension.length === 0) {\n return { perDimension, fidelity: Number.NaN, insufficientData, verdict: 'insufficient-data' }\n }\n const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length\n return {\n perDimension,\n fidelity,\n insufficientData,\n verdict: fidelity >= REPRESENTATIVE_MIN_FIDELITY ? 'representative' : 'skewed',\n }\n}\n\ntype FeatureValue = string | number | null\n\nfunction histograms(\n feature: string,\n simVals: FeatureValue[],\n prodVals: FeatureValue[],\n): { sim: Record<string, number>; prod: Record<string, number> } {\n const kinds = new Set<string>()\n for (const v of [...simVals, ...prodVals]) {\n if (v !== null) kinds.add(typeof v)\n }\n if (kinds.size > 1) {\n throw new ValidationError(\n `simFidelityReport: feature \"${feature}\" mixes string and number values — an extractor must return one kind per feature`,\n )\n }\n let toCategory: (v: string | number) => string\n if (kinds.has('number')) {\n const union: number[] = []\n for (const v of [...simVals, ...prodVals]) {\n if (v !== null) union.push(v as number)\n }\n const edges = quantileEdges(union)\n toCategory = (v) => bucketLabel(v as number, edges)\n } else {\n toCategory = (v) => v as string\n }\n const count = (vals: FeatureValue[]): Record<string, number> => {\n const hist: Record<string, number> = {}\n for (const v of vals) {\n const key = v === null ? ABSENT_CATEGORY : toCategory(v)\n hist[key] = (hist[key] ?? 0) + 1\n }\n return hist\n }\n return { sim: count(simVals), prod: count(prodVals) }\n}\n\nfunction topShifts(\n sim: Record<string, number>,\n simTotal: number,\n prod: Record<string, number>,\n prodTotal: number,\n): FeatureShift[] {\n const keys = [...new Set([...Object.keys(sim), ...Object.keys(prod)])]\n const shifts = keys.map((value) => ({\n value,\n pSim: (sim[value] ?? 0) / simTotal,\n pProd: (prod[value] ?? 0) / prodTotal,\n }))\n shifts.sort((a, b) => {\n const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd)\n return delta !== 0 ? delta : a.value.localeCompare(b.value)\n })\n return shifts.slice(0, TOP_SHIFT_COUNT)\n}\n\n// ── Easy-mode check ──────────────────────────────────────────────────\n\nexport interface EasyModeOptions {\n /** A run passes when its score (holdout, else search) >= this. Default 0.5\n * — matches the pass-threshold convention across the rl/ primitives. */\n passThreshold?: number\n /** Pass-rate gap above which the sim is flagged inflated. Default 0.1 —\n * a 10-point inflation is enough to flip most promotion gates. */\n inflationTolerance?: number\n}\n\nexport interface EasyModeReport {\n simPassRate: number\n prodPassRate: number\n /** simPassRate − prodPassRate. Positive = the simulator is easier than reality. */\n gap: number\n /** True when gap > inflationTolerance: numbers measured against this\n * simulator overstate production performance. */\n inflated: boolean\n}\n\n/**\n * The headline simulator failure mode as its own named artifact: a simulator\n * that creates \"easy mode\" inflates pass-rate relative to production, and\n * every score measured against it overstates reality. Throws on empty inputs\n * and on records carrying neither score — a silently-skipped record would\n * bias the very rate this check exists to keep honest.\n */\nexport function easyModeCheck(\n simulated: RunRecord[],\n production: RunRecord[],\n opts: EasyModeOptions = {},\n): EasyModeReport {\n if (simulated.length === 0) {\n throw new ValidationError('easyModeCheck: simulated records are empty')\n }\n if (production.length === 0) {\n throw new ValidationError('easyModeCheck: production records are empty')\n }\n const threshold = opts.passThreshold ?? 0.5\n const tolerance = opts.inflationTolerance ?? 0.1\n const passRate = (records: RunRecord[], side: string): number => {\n let passes = 0\n for (const r of records) {\n // RAW, same reason as `defaultBehaviorFeatures`: this rate exists to\n // catch a simulator that reports easier successes than production. Gating\n // would zero the inflated runs and hide the inflation being measured.\n const score =\n finiteOrNull(observedSplitScore(r, 'holdout')) ??\n finiteOrNull(observedSplitScore(r, 'search'))\n if (score === null) {\n throw new ValidationError(\n `easyModeCheck: ${side} run \"${r.runId}\" carries neither holdoutScore nor searchScore`,\n )\n }\n if (score >= threshold) passes++\n }\n return passes / records.length\n }\n const simPassRate = passRate(simulated, 'simulated')\n const prodPassRate = passRate(production, 'production')\n const gap = simPassRate - prodPassRate\n return { simPassRate, prodPassRate, gap, inflated: gap > tolerance }\n}\n","/**\n * Bradley-Terry / Elo tournament evaluation.\n *\n * For multi-candidate sweeps, comparing every candidate's score against\n * a fixed comparator wastes information — the comparator becomes a high-\n * variance reference and rank flips between near-tied middle-rank\n * candidates are dominated by noise. Pairwise tournaments fix this:\n * every (i, j) pair contributes a comparison to a Bradley-Terry MLE that\n * estimates each candidate's strength on a unified scale.\n *\n * For online updating (rolling campaigns where new candidates arrive\n * over time), we also ship classical Elo with configurable K-factor.\n *\n * References:\n * - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete\n * block designs. Biometrika, 39(3/4), 324–345.\n * - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry\n * models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm\n * used here.)\n * - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.\n *\n * This is a useful primitive because most LLM-eval communities (Chatbot\n * Arena, AlpacaEval, ELO-style ablation) have converged on pairwise\n * tournament eval as the most sample-efficient and most rank-stable\n * method when you have many candidates.\n */\n\nexport interface PairwiseOutcome {\n /** Winner candidate id. */\n winner: string\n /** Loser candidate id. */\n loser: string\n /**\n * Optional draw flag. When true, both candidates get half-credit\n * (Bradley-Terry handles draws as half-wins for each side).\n */\n draw?: boolean\n /**\n * Optional weight — useful if some pairwise comparisons are stronger\n * signals than others (e.g. a paired test with a wider score gap is\n * a more confident comparison). Default 1.\n */\n weight?: number\n}\n\nexport interface BradleyTerryRating {\n candidateId: string\n /** Latent strength θ ≥ 0 from the BT MLE. */\n strength: number\n /** Log-strength = log(θ) — interpretable on a linear scale. */\n logStrength: number\n /** Number of pairwise comparisons this candidate appears in. */\n n: number\n /** Win count (+ 0.5 per draw). */\n wins: number\n}\n\nexport interface BradleyTerryFit {\n ratings: BradleyTerryRating[]\n /** Iterations of the MM algorithm before convergence. */\n iterations: number\n /** Final maximum |θ_new - θ_old| / θ_old. */\n finalDelta: number\n converged: boolean\n}\n\n/**\n * Bradley-Terry MLE via Hunter's MM algorithm.\n *\n * Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)\n * where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.\n *\n * Returns log-strengths normalized so the smallest is 0 (any constant\n * offset is unobservable in BT — only differences are identified).\n */\nexport function fitBradleyTerry(\n outcomes: PairwiseOutcome[],\n opts: { tolerance?: number; maxIterations?: number; smoothing?: number } = {},\n): BradleyTerryFit {\n const tol = opts.tolerance ?? 1e-6\n const maxIter = opts.maxIterations ?? 256\n // Small positive default — Hunter's MM degenerates when a candidate has\n // zero wins (θ → 0 → log → -∞). 0.1 is negligible against real win counts\n // (~1 win / 10 comparisons) and keeps the iteration well-conditioned.\n // Override to 0 if the comparison set is guaranteed strongly connected.\n const smoothing = opts.smoothing ?? 0.1\n\n const candidates = new Set<string>()\n for (const o of outcomes) {\n candidates.add(o.winner)\n candidates.add(o.loser)\n }\n const ids = [...candidates].sort()\n const idx = new Map(ids.map((id, i) => [id, i]))\n const n = ids.length\n if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true }\n if (n === 1) {\n return {\n ratings: [{ candidateId: ids[0]!, strength: 1, logStrength: 0, n: 0, wins: 0 }],\n iterations: 0,\n finalDelta: 0,\n converged: true,\n }\n }\n\n // Build win matrix W[i][j] = (weighted) wins of i over j, plus half for draws.\n // Build comparison matrix N[i][j] = total weighted comparisons between i and j.\n const W: number[][] = Array.from({ length: n }, () => new Array<number>(n).fill(0))\n const N: number[][] = Array.from({ length: n }, () => new Array<number>(n).fill(0))\n for (const o of outcomes) {\n const i = idx.get(o.winner)!\n const j = idx.get(o.loser)!\n const w = o.weight ?? 1\n if (o.draw) {\n W[i]![j]! += 0.5 * w\n W[j]![i]! += 0.5 * w\n } else {\n W[i]![j]! += w\n }\n N[i]![j]! += w\n N[j]![i]! += w\n }\n\n // Per-candidate total wins.\n const winsTotal = new Array<number>(n).fill(0)\n for (let i = 0; i < n; i++) {\n for (let j = 0; j < n; j++) winsTotal[i]! += W[i]![j]!\n winsTotal[i]! += smoothing // tiny smoothing to keep θ positive\n }\n const compsTotal = new Array<number>(n).fill(0)\n for (let i = 0; i < n; i++) {\n for (let j = 0; j < n; j++) compsTotal[i]! += N[i]![j]!\n }\n\n // MM iterations.\n let theta = new Array<number>(n).fill(1)\n let iter = 0\n let delta = Infinity\n for (; iter < maxIter; iter++) {\n const newTheta = new Array<number>(n)\n for (let i = 0; i < n; i++) {\n let denom = 0\n for (let j = 0; j < n; j++) {\n if (j === i) continue\n if (N[i]![j]! === 0) continue\n denom += N[i]![j]! / (theta[i]! + theta[j]!)\n }\n newTheta[i] = denom === 0 ? theta[i]! : winsTotal[i]! / denom\n }\n // Normalize so geometric mean = 1 (numerical stability).\n let logSum = 0\n for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]!))\n const norm = Math.exp(logSum / n)\n for (let i = 0; i < n; i++) newTheta[i] = newTheta[i]! / norm\n\n delta = 0\n for (let i = 0; i < n; i++) {\n const d = Math.abs(newTheta[i]! - theta[i]!) / Math.max(1e-12, theta[i]!)\n if (d > delta) delta = d\n }\n theta = newTheta\n if (delta < tol) break\n }\n\n const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))))\n const ratings: BradleyTerryRating[] = ids.map((id, i) => ({\n candidateId: id,\n strength: theta[i]!,\n logStrength: Math.log(Math.max(1e-300, theta[i]!)) - minLog,\n n: compsTotal[i]!,\n wins: winsTotal[i]! - smoothing,\n }))\n\n return {\n ratings: ratings.sort((a, b) => b.strength - a.strength),\n iterations: iter,\n finalDelta: delta,\n converged: delta < tol,\n }\n}\n\n/**\n * Online Elo updates. Use when comparisons arrive over time and you want\n * a running rating without re-fitting the full BT MLE on every update.\n *\n * Initialize ratings to `defaultRating` (1500 by default). Each call to\n * `applyEloUpdate` mutates the map in place and returns the deltas so\n * the caller can log per-comparison rating changes.\n */\nexport interface EloOptions {\n /** Default rating for unseen candidates. Default 1500. */\n defaultRating?: number\n /** K-factor controls the step size. Default 32 (FIDE-ish). */\n kFactor?: number\n}\n\nexport function applyEloUpdate(\n ratings: Map<string, number>,\n outcome: PairwiseOutcome,\n opts: EloOptions = {},\n): { winnerDelta: number; loserDelta: number } {\n const defaultRating = opts.defaultRating ?? 1500\n const k = opts.kFactor ?? 32\n\n const rW = ratings.get(outcome.winner) ?? defaultRating\n const rL = ratings.get(outcome.loser) ?? defaultRating\n\n const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400))\n const scoreW = outcome.draw ? 0.5 : 1\n const scoreL = outcome.draw ? 0.5 : 0\n const w = outcome.weight ?? 1\n\n const winnerDelta = k * w * (scoreW - expectedW)\n const loserDelta = k * w * (scoreL - (1 - expectedW))\n\n ratings.set(outcome.winner, rW + winnerDelta)\n ratings.set(outcome.loser, rL + loserDelta)\n\n return { winnerDelta, loserDelta }\n}\n\n/**\n * Build pairwise outcomes from the campaign artifact: for every scenario\n * shared by two candidates, the higher-scoring run wins. Useful when you\n * want a tournament view of an existing campaign without an additional\n * pairwise judge call.\n */\nexport interface BuildPairwiseFromCampaignInput {\n runs: Array<{\n candidateId: string\n /** Stable identifier for the matching unit (typically scenarioId). */\n matchKey: string\n score: number\n }>\n /**\n * Tied-score margin. Below this, the comparison is a draw. Default 0\n * (no ties).\n */\n drawMargin?: number\n}\n\nexport function buildPairwiseFromCampaign(\n input: BuildPairwiseFromCampaignInput,\n): PairwiseOutcome[] {\n const drawMargin = input.drawMargin ?? 0\n const byKey = new Map<string, Array<{ candidateId: string; score: number }>>()\n for (const r of input.runs) {\n const arr = byKey.get(r.matchKey) ?? []\n arr.push({ candidateId: r.candidateId, score: r.score })\n byKey.set(r.matchKey, arr)\n }\n const outcomes: PairwiseOutcome[] = []\n for (const arr of byKey.values()) {\n for (let i = 0; i < arr.length; i++) {\n for (let j = i + 1; j < arr.length; j++) {\n const a = arr[i]!\n const b = arr[j]!\n if (a.candidateId === b.candidateId) continue\n const margin = Math.abs(a.score - b.score)\n if (margin <= drawMargin) {\n outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 })\n } else {\n const [winner, loser] = a.score > b.score ? [a, b] : [b, a]\n outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin })\n }\n }\n }\n }\n return outcomes\n}\n","/**\n * Verified-findings dataset — execution-verified gold labels as RL-ready rows.\n *\n * A replay-verify batch re-executes a labeled trajectory prefix inside the\n * original docker image and checks, at the gold \"incorrect\" step k, whether\n * the recorded failure reproduces (arm A) and whether a generated fix makes\n * it vanish (arm B). That turns an annotation into an *executed* label: the\n * verdict is a returncode/signature comparison, not a rater's opinion.\n *\n * This module joins three artifact families into one row per replayed case:\n *\n * 1. the batch report (`batch-report.json` — per-case verdicts, fix arms),\n * 2. the gold label corpus (`*-labels.json` — incorrect step annotations),\n * 3. the normalized trajectory (`normalized/<trajId>/steps.json` — the\n * action/observation sequence the agent actually took).\n *\n * The emitted `VerifiedFindingRow` carries the trajectory prefix up to k,\n * the gold label, the execution verdict with its evidence (exit codes,\n * failure signature, prefix divergences), the fix arm when present, and\n * per-row provenance (label/steps/report sha256s, docker images, run ids).\n * Rows are trainer input for step-level localizer/critic models; the reward\n * is deterministic because execution decided it.\n *\n * Join discipline: every missing or inconsistent join throws — a dataset\n * built from partially joined artifacts would silently train on wrong\n * labels. The batch report is authoritative for fix outcomes (per-case\n * `replay-verdict.json` files are written before the fix arm completes);\n * per-case files contribute prefix-divergence detail and run ids only, and\n * are cross-checked against the report where they overlap.\n */\n\nimport { createHash } from 'node:crypto'\nimport { readFileSync } from 'node:fs'\nimport { join } from 'node:path'\nimport { compareCodeUnits } from '../ledger-core/canonical'\n\nexport const VERIFIED_FINDING_SCHEMA = 'agent-eval/verified-finding@0'\n\n// ── Input shapes (parsed artifacts) ─────────────────────────────────\n\n/** One case row from a replay-verify `batch-report.json`. */\nexport interface ReplayBatchCase {\n corpus: string\n trajId: string\n image: string\n cwd: string\n cwdSource: string\n k: number\n stepCount: number\n goldIncorrectSteps: number[]\n recordedReturncodeAtK: number\n derivedImage: string | null\n signature: string | null\n status: string\n error: string | null\n prefixExecuted: number\n prefixDivergences: number\n prefixDivergencePct: number\n prefixReturncodeMismatches: number\n prefixUnknownExpectations: number\n armAExit: number | null\n armAReturncodeMatch: boolean\n armASignatureMatch: boolean\n /** Batch verdict: prefix divergence within tolerance AND arm A reproduced the recorded returncode at k. */\n replayed: boolean\n fix: ReplayBatchFix | null\n wallMs: number\n}\n\nexport interface ReplayBatchFix {\n attempted: boolean\n sampledOut: boolean\n command: string | null\n llmError: string | null\n armBExit: number | null\n failureVanished: boolean | null\n}\n\nexport interface ReplayBatchReport {\n generatedAt: string\n cases: ReplayBatchCase[]\n}\n\n/** Gold label entry for one trajectory (CodeTraceBench annotation format). */\nexport interface GoldLabelEntry {\n traj_id: string\n solved: boolean\n step_count: number\n agent?: string\n model?: string\n task_name?: string\n difficulty?: string\n incorrect_stages: Array<{ stage_id: number; incorrect_step_ids: number[] }>\n}\n\n/** One step from a normalized trajectory `steps.json`. */\nexport interface NormalizedStep {\n step_id: number\n action: string\n observation?: string | null\n}\n\nexport interface PrefixDivergence {\n step: number\n /** `unknown-expectation` marks a step the recording carries no returncode\n * for: it could not be confirmed, so it is not agreement. */\n kind: 'returncode-mismatch' | 'unknown-expectation'\n /** null exactly when `kind` is `unknown-expectation`. */\n expectedReturncode: number | null\n actualExit: number\n}\n\n/** Optional extract from a per-case `replay-verdict.json` (arm A detail only). */\nexport interface CaseVerdictDetail {\n k: number\n prefixExecuted: number\n recordedReturncode: number\n signatureBasis: string | null\n prefixDivergences: PrefixDivergence[]\n armACommand: string | null\n runIds: { original: string | null; armA: string | null }\n}\n\n// ── Row schema ──────────────────────────────────────────────────────\n\nexport interface TrajectoryStep {\n stepId: number\n action: string\n observation: string | null\n /** True when the observation was cut at `maxObservationChars`; `observationChars` keeps the original length. */\n observationTruncated: boolean\n observationChars: number\n}\n\nexport type FixOutcome = 'flipped' | 'not-flipped' | 'generation-failed' | 'not-attempted'\n\nexport interface VerifiedFindingFix {\n outcome: FixOutcome\n command: string | null\n llmError: string | null\n armBExit: number | null\n failureVanished: boolean | null\n}\n\nexport interface VerifiedFindingRow {\n schema: typeof VERIFIED_FINDING_SCHEMA\n /** `<runId>/<corpus>/<trajId>` — unique across batches. */\n caseId: string\n corpus: string\n trajId: string\n task: {\n agent: string | null\n model: string | null\n taskName: string | null\n difficulty: string | null\n solved: boolean\n stepCount: number\n }\n gold: {\n /** The verified gold step — the earliest replayable incorrect step. */\n stepK: number\n /** The exact command the agent ran at step k (never truncated — it is the labeled object). */\n actionAtK: string\n /** Incorrect steps the batch considered replay targets (submit-step golds excluded). */\n goldIncorrectSteps: number[]\n /** Every incorrect step in the label entry, across stages. */\n labelIncorrectSteps: number[]\n recordedReturncodeAtK: number\n }\n /** Prefix context 1..k — post-k steps are excluded so a trainer never sees the future. */\n trajectory: {\n window: { start: number; end: number }\n steps: TrajectoryStep[]\n }\n verification: {\n reproduced: boolean\n /** Arm A output also contained the recorded error substring (or returncode-only basis matched). */\n signatureStrict: boolean\n signatureBasis: string | null\n signature: string | null\n prefixExecuted: number\n prefixDivergences: number\n prefixDivergencePct: number\n prefixReturncodeMismatches: number\n /** Steps the recording carries no returncode for. Nonzero here means part\n * of the prefix replay was never confirmed against the recording. */\n prefixUnknownExpectations: number\n prefixDivergenceDetail: PrefixDivergence[] | null\n armAExit: number | null\n armAReturncodeMatch: boolean\n armACommand: string | null\n wallMs: number\n }\n fix: VerifiedFindingFix\n provenance: {\n runId: string\n batchGeneratedAt: string\n batchReportSha256: string\n labelsPath: string\n labelsSha256: string\n stepsPath: string\n stepsSha256: string\n image: string\n derivedImage: string | null\n cwd: string\n cwdSource: string\n originalRunId: string | null\n armARunId: string | null\n }\n}\n\nexport interface VerifiedFindingsSummary {\n rows: number\n reproduced: number\n /** Reproduced AND arm A matched the failure signature — the batch report's headline strict rate.\n * Row-level `verification.signatureStrict` is raw arm A evidence and can be true on a\n * non-reproduced case (signature matched but the prefix diverged past tolerance). */\n signatureStrict: number\n fix: Record<FixOutcome, number>\n byCorpus: Record<string, { rows: number; reproduced: number; fixFlipped: number }>\n}\n\n// ── Pure join ───────────────────────────────────────────────────────\n\nconst DEFAULT_MAX_OBSERVATION_CHARS = 4000\n\nexport interface BuildVerifiedFindingRowArgs {\n batchCase: ReplayBatchCase\n label: GoldLabelEntry\n steps: NormalizedStep[]\n runId: string\n batchGeneratedAt: string\n batchReportSha256: string\n labelsPath: string\n labelsSha256: string\n stepsPath: string\n stepsSha256: string\n detail?: CaseVerdictDetail\n maxObservationChars?: number\n}\n\nfunction fail(caseId: string, message: string): never {\n throw new Error(`verified-findings: ${caseId}: ${message}`)\n}\n\nfunction deriveFixOutcome(caseId: string, batchCase: ReplayBatchCase): VerifiedFindingFix {\n const fix = batchCase.fix\n if (fix === null) {\n if (batchCase.replayed) {\n fail(\n caseId,\n 'replayed case has no fix record — the batch always records the fix arm for replayed cases',\n )\n }\n return {\n outcome: 'not-attempted',\n command: null,\n llmError: null,\n armBExit: null,\n failureVanished: null,\n }\n }\n const base = {\n command: fix.command,\n llmError: fix.llmError,\n armBExit: fix.armBExit,\n failureVanished: fix.failureVanished,\n }\n if (fix.command !== null) {\n if (fix.failureVanished === null) {\n fail(\n caseId,\n 'fix command present but failureVanished missing — arm B verdict was never recorded',\n )\n }\n return { outcome: fix.failureVanished ? 'flipped' : 'not-flipped', ...base }\n }\n if (fix.llmError !== null) return { outcome: 'generation-failed', ...base }\n if (!fix.attempted || fix.sampledOut) return { outcome: 'not-attempted', ...base }\n fail(caseId, 'unrecognized fix record state (attempted, no command, no llmError)')\n}\n\nfunction truncateObservation(\n observation: string | null | undefined,\n maxChars: number,\n): Pick<TrajectoryStep, 'observation' | 'observationTruncated' | 'observationChars'> {\n if (observation === null || observation === undefined) {\n return { observation: null, observationTruncated: false, observationChars: 0 }\n }\n if (observation.length <= maxChars) {\n return { observation, observationTruncated: false, observationChars: observation.length }\n }\n return {\n observation: observation.slice(0, maxChars),\n observationTruncated: true,\n observationChars: observation.length,\n }\n}\n\n/**\n * Join one batch case with its gold label and trajectory into a row.\n * Throws on any join inconsistency — never emits a partially joined row.\n */\nexport function buildVerifiedFindingRow(args: BuildVerifiedFindingRowArgs): VerifiedFindingRow {\n const { batchCase, label, steps, detail } = args\n const caseId = `${args.runId}/${batchCase.corpus}/${batchCase.trajId}`\n const maxObservationChars = args.maxObservationChars ?? DEFAULT_MAX_OBSERVATION_CHARS\n\n if (batchCase.status !== 'ok') {\n fail(\n caseId,\n `case status is '${batchCase.status}' (error: ${batchCase.error ?? 'none'}) — only ok cases join`,\n )\n }\n if (label.traj_id !== batchCase.trajId) {\n fail(caseId, `label traj_id '${label.traj_id}' does not match the case`)\n }\n if (label.step_count !== batchCase.stepCount) {\n fail(caseId, `label step_count ${label.step_count} != case stepCount ${batchCase.stepCount}`)\n }\n if (steps.length !== batchCase.stepCount) {\n fail(caseId, `steps.json has ${steps.length} steps, case expects ${batchCase.stepCount}`)\n }\n for (let i = 0; i < steps.length; i++) {\n const step = steps[i]!\n if (step.step_id !== i + 1) {\n fail(caseId, `steps.json is not contiguous 1..n: index ${i} has step_id ${step.step_id}`)\n }\n }\n const k = batchCase.k\n if (k < 1 || k > batchCase.stepCount) {\n fail(caseId, `gold step k=${k} is outside 1..${batchCase.stepCount}`)\n }\n if (!batchCase.goldIncorrectSteps.includes(k)) {\n fail(\n caseId,\n `gold step k=${k} is not in goldIncorrectSteps [${batchCase.goldIncorrectSteps.join(', ')}]`,\n )\n }\n const labelIncorrectSteps = [\n ...new Set(label.incorrect_stages.flatMap((s) => s.incorrect_step_ids)),\n ].sort((a, b) => a - b)\n for (const goldStep of batchCase.goldIncorrectSteps) {\n if (!labelIncorrectSteps.includes(goldStep)) {\n fail(\n caseId,\n `case gold step ${goldStep} is absent from the label's incorrect steps — label/report mismatch`,\n )\n }\n }\n if (detail !== undefined) {\n if (detail.k !== k) fail(caseId, `per-case verdict k=${detail.k} != report k=${k}`)\n if (detail.prefixExecuted !== batchCase.prefixExecuted) {\n fail(\n caseId,\n `per-case verdict prefixExecuted=${detail.prefixExecuted} != report ${batchCase.prefixExecuted}`,\n )\n }\n if (detail.recordedReturncode !== batchCase.recordedReturncodeAtK) {\n fail(\n caseId,\n `per-case verdict recordedReturncode=${detail.recordedReturncode} != report ${batchCase.recordedReturncodeAtK}`,\n )\n }\n if (detail.prefixDivergences.length !== batchCase.prefixDivergences) {\n fail(\n caseId,\n `per-case verdict lists ${detail.prefixDivergences.length} prefix divergences != report ${batchCase.prefixDivergences}`,\n )\n }\n const unknown = detail.prefixDivergences.filter((d) => d.kind === 'unknown-expectation').length\n if (unknown !== batchCase.prefixUnknownExpectations) {\n fail(\n caseId,\n `per-case verdict lists ${unknown} unknown-expectation steps != report ${batchCase.prefixUnknownExpectations}`,\n )\n }\n }\n\n const stepAtK = steps[k - 1]!\n const trajectorySteps: TrajectoryStep[] = steps.slice(0, k).map((step) => ({\n stepId: step.step_id,\n action: step.action,\n ...truncateObservation(step.observation, maxObservationChars),\n }))\n\n return {\n schema: VERIFIED_FINDING_SCHEMA,\n caseId,\n corpus: batchCase.corpus,\n trajId: batchCase.trajId,\n task: {\n agent: label.agent ?? null,\n model: label.model ?? null,\n taskName: label.task_name ?? null,\n difficulty: label.difficulty ?? null,\n solved: label.solved,\n stepCount: batchCase.stepCount,\n },\n gold: {\n stepK: k,\n actionAtK: stepAtK.action,\n goldIncorrectSteps: [...batchCase.goldIncorrectSteps].sort((a, b) => a - b),\n labelIncorrectSteps,\n recordedReturncodeAtK: batchCase.recordedReturncodeAtK,\n },\n trajectory: {\n window: { start: 1, end: k },\n steps: trajectorySteps,\n },\n verification: {\n reproduced: batchCase.replayed,\n signatureStrict: batchCase.armASignatureMatch,\n signatureBasis: detail?.signatureBasis ?? null,\n signature: batchCase.signature,\n prefixExecuted: batchCase.prefixExecuted,\n prefixDivergences: batchCase.prefixDivergences,\n prefixDivergencePct: batchCase.prefixDivergencePct,\n prefixReturncodeMismatches: batchCase.prefixReturncodeMismatches,\n prefixUnknownExpectations: batchCase.prefixUnknownExpectations,\n prefixDivergenceDetail: detail?.prefixDivergences ?? null,\n armAExit: batchCase.armAExit,\n armAReturncodeMatch: batchCase.armAReturncodeMatch,\n armACommand: detail?.armACommand ?? null,\n wallMs: batchCase.wallMs,\n },\n fix: deriveFixOutcome(caseId, batchCase),\n provenance: {\n runId: args.runId,\n batchGeneratedAt: args.batchGeneratedAt,\n batchReportSha256: args.batchReportSha256,\n labelsPath: args.labelsPath,\n labelsSha256: args.labelsSha256,\n stepsPath: args.stepsPath,\n stepsSha256: args.stepsSha256,\n image: batchCase.image,\n derivedImage: batchCase.derivedImage,\n cwd: batchCase.cwd,\n cwdSource: batchCase.cwdSource,\n originalRunId: detail?.runIds.original ?? null,\n armARunId: detail?.runIds.armA ?? null,\n },\n }\n}\n\nexport function summarizeVerifiedFindings(rows: VerifiedFindingRow[]): VerifiedFindingsSummary {\n const summary: VerifiedFindingsSummary = {\n rows: rows.length,\n reproduced: 0,\n signatureStrict: 0,\n fix: { flipped: 0, 'not-flipped': 0, 'generation-failed': 0, 'not-attempted': 0 },\n byCorpus: {},\n }\n for (const row of rows) {\n if (row.verification.reproduced) summary.reproduced++\n if (row.verification.reproduced && row.verification.signatureStrict) summary.signatureStrict++\n summary.fix[row.fix.outcome]++\n let corpus = summary.byCorpus[row.corpus]\n if (corpus === undefined) {\n corpus = { rows: 0, reproduced: 0, fixFlipped: 0 }\n summary.byCorpus[row.corpus] = corpus\n }\n corpus.rows++\n if (row.verification.reproduced) corpus.reproduced++\n if (row.fix.outcome === 'flipped') corpus.fixFlipped++\n }\n return summary\n}\n\nexport function verifiedFindingsToJsonl(rows: VerifiedFindingRow[]): string {\n return rows.map((row) => JSON.stringify(row)).join('\\n') + (rows.length > 0 ? '\\n' : '')\n}\n\n// ── Filesystem loader ───────────────────────────────────────────────\n\nexport interface VerifiedFindingsCorpusSource {\n labelsPath: string\n /** Directory containing `normalized/<trajId>/steps.json`. */\n preparedDir: string\n}\n\nexport interface VerifiedFindingsSource {\n batchReportPath: string\n /** Batch run identifier embedded in every caseId, e.g. 'run2-20260802'. */\n runId: string\n /** Corpus name (as it appears in the batch report) → label + trajectory locations. */\n corpora: Record<string, VerifiedFindingsCorpusSource>\n /** Batch run directory holding `<corpus>--<trajId>/replay-verdict.json`; when set, every case must have one. */\n runDir?: string\n maxObservationChars?: number\n}\n\nexport interface VerifiedFindingsDataset {\n rows: VerifiedFindingRow[]\n summary: VerifiedFindingsSummary\n provenance: {\n runId: string\n batchReportPath: string\n batchReportSha256: string\n batchGeneratedAt: string\n corpora: Record<string, { labelsPath: string; labelsSha256: string; preparedDir: string }>\n }\n}\n\nfunction sha256(buffer: Buffer): string {\n return createHash('sha256').update(buffer).digest('hex')\n}\n\nfunction readJson(path: string, what: string): { value: unknown; sha256: string } {\n let buffer: Buffer\n try {\n buffer = readFileSync(path)\n } catch (error) {\n throw new Error(\n `verified-findings: cannot read ${what} at ${path}: ${(error as Error).message}`,\n )\n }\n try {\n return { value: JSON.parse(buffer.toString('utf8')), sha256: sha256(buffer) }\n } catch (error) {\n throw new Error(\n `verified-findings: ${what} at ${path} is not valid JSON: ${(error as Error).message}`,\n )\n }\n}\n\ninterface CaseVerdictFile {\n k: number\n prefixExecuted: number\n recordedReturncode: number\n signatureBasis?: string | null\n prefixDivergences?: PrefixDivergence[]\n armA?: { command?: string | null } | null\n runIds?: { original?: string | null; armA?: string | null } | null\n}\n\nconst PREFIX_DIVERGENCE_KINDS = new Set(['returncode-mismatch', 'unknown-expectation'])\n\nfunction loadCaseVerdictDetail(runDir: string, batchCase: ReplayBatchCase): CaseVerdictDetail {\n const path = join(runDir, `${batchCase.corpus}--${batchCase.trajId}`, 'replay-verdict.json')\n const parsed = readJson(path, `per-case verdict for ${batchCase.trajId}`).value as CaseVerdictFile\n const divergences = parsed.prefixDivergences ?? []\n // A record without a kind came from a comparison that could not tell a\n // confirmed match from an unverifiable step, so the whole file is unusable.\n for (const divergence of divergences) {\n if (!PREFIX_DIVERGENCE_KINDS.has(divergence.kind)) {\n throw new Error(\n `verified-findings: ${path} step ${divergence.step} has divergence kind ` +\n `'${divergence.kind}'; expected one of ${[...PREFIX_DIVERGENCE_KINDS].join(', ')}`,\n )\n }\n }\n return {\n k: parsed.k,\n prefixExecuted: parsed.prefixExecuted,\n recordedReturncode: parsed.recordedReturncode,\n signatureBasis: parsed.signatureBasis ?? null,\n prefixDivergences: divergences,\n armACommand: parsed.armA?.command ?? null,\n runIds: {\n original: parsed.runIds?.original ?? null,\n armA: parsed.runIds?.armA ?? null,\n },\n }\n}\n\n/**\n * Load a replay-verify batch and join it into verified-finding rows.\n * Every case in the report must join: an unresolvable corpus, a missing\n * label entry, or a missing trajectory throws instead of dropping the row.\n */\nexport function loadVerifiedFindingsDataset(\n source: VerifiedFindingsSource,\n): VerifiedFindingsDataset {\n const report = readJson(source.batchReportPath, 'batch report')\n const parsedReport = report.value as ReplayBatchReport\n if (!Array.isArray(parsedReport.cases) || parsedReport.cases.length === 0) {\n throw new Error(`verified-findings: batch report at ${source.batchReportPath} has no cases`)\n }\n if (typeof parsedReport.generatedAt !== 'string' || parsedReport.generatedAt.length === 0) {\n throw new Error(\n `verified-findings: batch report at ${source.batchReportPath} has no generatedAt`,\n )\n }\n\n const labelCache = new Map<string, { sha256: string; byTrajId: Map<string, GoldLabelEntry> }>()\n const corporaProvenance: VerifiedFindingsDataset['provenance']['corpora'] = {}\n\n const resolveCorpus = (corpus: string) => {\n const config = source.corpora[corpus]\n if (config === undefined) {\n throw new Error(\n `verified-findings: batch report references corpus '${corpus}' but no labels/preparedDir was configured for it`,\n )\n }\n let cached = labelCache.get(corpus)\n if (cached === undefined) {\n const labels = readJson(config.labelsPath, `labels for corpus '${corpus}'`)\n const entries = labels.value as GoldLabelEntry[]\n if (!Array.isArray(entries)) {\n throw new Error(\n `verified-findings: labels for corpus '${corpus}' at ${config.labelsPath} are not an array`,\n )\n }\n const byTrajId = new Map<string, GoldLabelEntry>()\n for (const entry of entries) {\n if (byTrajId.has(entry.traj_id)) {\n throw new Error(\n `verified-findings: labels for corpus '${corpus}' contain duplicate traj_id '${entry.traj_id}'`,\n )\n }\n byTrajId.set(entry.traj_id, entry)\n }\n cached = { sha256: labels.sha256, byTrajId }\n labelCache.set(corpus, cached)\n corporaProvenance[corpus] = {\n labelsPath: config.labelsPath,\n labelsSha256: labels.sha256,\n preparedDir: config.preparedDir,\n }\n }\n return { config, ...cached }\n }\n\n const rows: VerifiedFindingRow[] = []\n for (const batchCase of parsedReport.cases) {\n const { config, sha256: labelsSha256, byTrajId } = resolveCorpus(batchCase.corpus)\n const label = byTrajId.get(batchCase.trajId)\n if (label === undefined) {\n throw new Error(\n `verified-findings: ${source.runId}/${batchCase.corpus}/${batchCase.trajId}: no label entry in ${config.labelsPath}`,\n )\n }\n const stepsPath = join(config.preparedDir, 'normalized', batchCase.trajId, 'steps.json')\n const stepsFile = readJson(stepsPath, `trajectory steps for ${batchCase.trajId}`)\n const steps = stepsFile.value as NormalizedStep[]\n if (!Array.isArray(steps)) {\n throw new Error(`verified-findings: trajectory steps at ${stepsPath} are not an array`)\n }\n const detail =\n source.runDir === undefined ? undefined : loadCaseVerdictDetail(source.runDir, batchCase)\n rows.push(\n buildVerifiedFindingRow({\n batchCase,\n label,\n steps,\n runId: source.runId,\n batchGeneratedAt: parsedReport.generatedAt,\n batchReportSha256: report.sha256,\n labelsPath: config.labelsPath,\n labelsSha256,\n stepsPath,\n stepsSha256: stepsFile.sha256,\n detail,\n maxObservationChars: source.maxObservationChars,\n }),\n )\n }\n rows.sort((a, b) => compareCodeUnits(a.caseId, b.caseId))\n\n return {\n rows,\n summary: summarizeVerifiedFindings(rows),\n provenance: {\n runId: source.runId,\n batchReportPath: source.batchReportPath,\n batchReportSha256: report.sha256,\n batchGeneratedAt: parsedReport.generatedAt,\n corpora: corporaProvenance,\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2EA,eAAsB,mBACpB,MAC0B;CAC1B,MAAM,KAAK,KAAK,MAAM;EAAC;EAAG;EAAG;EAAG;EAAG;EAAG;CAAE;CACxC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,WAAW,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAE7C,MAAM,SAA4B,CAAC;CACnC,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,cAA8C,CAAC;EACrD,MAAM,YAAsB,CAAC;EAC7B,IAAI,cAAc;EAClB,IAAI,gBAAgB;EACpB,KAAK,MAAM,YAAY,KAAK,WAAW;GACrC,MAAM,MAAM,SAAS,cAAc,YAAY,KAAK,UAAU,QAAQ,QAAQ;GAC9E,MAAM,SAAmB,CAAC;GAC1B,IAAI,SAAS;GACb,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,KAAK;IAC7B,MAAM,QAAQ,MAAM,KAAK,OAAO,IAAI;KAAE;KAAU;KAAG,KAAK;IAAE,CAAC;IAC3D,OAAO,KAAK,KAAK;IACjB,IAAI,SAAS,eAAe;IAC5B,UAAU,KAAK,KAAK;IACpB,IAAI,SAAS,eAAe;IAC5B;GACF;GACA,MAAM,QAAQ,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;GACzD,YAAY,KAAK;IAAE,YAAY;IAAK,WAAW;IAAO;IAAQ,OAAO,OAAO;GAAO,CAAC;EACtF;EACA,MAAM,YAAY,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,UAAU,MAAM;EACrF,MAAM,WACJ,UAAU,SAAS,IACf,IACA,UAAU,QAAQ,GAAG,MAAM,KAAK,IAAI,cAAc,GAAG,CAAC,KAAK,UAAU,SAAS;EACpF,OAAO,KAAK;GACV;GACA;GACA,UAAU,cAAc,KAAK,IAAI,GAAG,aAAa;GACjD,KAAK,KAAK,KAAK,QAAQ;GACvB,GAAG,UAAU;GACb;EACF,CAAC;CACH;CAEA,MAAM,aAAa,OAAO,MAAM,MAAM,EAAE,YAAY,aAAa,CAAC,EAAE,KAAK;CACzE,MAAM,OAAO,SAAS,SAAS,SAAS,MAAM;CAE9C,IAAI,OAAO;CACX,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;EACtC,MAAM,KAAK,OAAO,IAAI,EAAE,CAAE;EAC1B,MAAM,KAAK,OAAO,EAAE,CAAE;EACtB,MAAM,KAAK,OAAO,IAAI,EAAE,CAAE;EAC1B,MAAM,KAAK,OAAO,EAAE,CAAE;EACtB,SAAU,KAAK,MAAM,KAAM,KAAK;CAClC;CAGA,OAAO;EAAE;EAAQ;EAAY,gBAFN,SAAS,IAAI,IAAI,OAAO;CAEH;AAC9C;;;;;;AAwBA,SAAgB,wBACd,GACA,GACA,OAA4E,CAAC,GACxD;CACrB,MAAM,OAAO,KAAK,cAAc;CAChC,MAAM,YAAY,KAAK,sBAAsB;CAC7C,MAAM,MAAM,QACV,KAAK,MACL,EAAE,OAAO,SAAS,UAAU,MAAM,YAAY,KAAK,SAAS,KAAK,SAAS,CAAC,GAC3E,EAAE,OAAO,SAAS,UAAU,MAAM,YAAY,KAAK,SAAS,KAAK,SAAS,CAAC,CAC7E;CAEA,MAAM,OAAoC,CAAC;CAC3C,KAAK,MAAM,MAAM,EAAE,QAAQ;EACzB,MAAM,KAAK,EAAE,OAAO,MAAM,MAAM,EAAE,MAAM,GAAG,CAAC;EAC5C,IAAI,CAAC,IAAI;EACT,MAAM,SAAS,GAAG,YAAY,KAAK,MAAM,EAAE,SAAS;EACpD,MAAM,SAAS,GAAG,YAAY,KAAK,MAAM,EAAE,SAAS;EACpD,MAAM,MAAM,gBAAgB,QAAQ,WAAW,MAAM,GAAG;EACxD,MAAM,MAAM,gBAAgB,QAAQ,WAAW,MAAM,GAAG;EACxD,KAAK,KAAK;GACR,GAAG,GAAG;GACN,WAAW,GAAG,YAAY,GAAG;GAC7B,MAAM,IAAI;GACV,OAAO,IAAI;GACX,MAAM,IAAI;GACV,OAAO,IAAI;EACb,CAAC;CACH;CAEA,MAAM,YAAY,EAAE,iBAAiB,EAAE;CACvC,MAAM,kBACJ,EAAE,eAAe,QAAQ,EAAE,eAAe,OACtC,EAAE,aAAa,EAAE,aACjB;CAIN,MAAM,YAAY,KAAK,QAAQ,GAAG,MAAM,IAAI,EAAE,WAAW,CAAC,IAAI,KAAK,IAAI,GAAG,KAAK,MAAM;CACrF,IAAI;CACJ,IAAI,KAAK,IAAI,SAAS,IAAI,OAAQ,KAAK,IAAI,SAAS,IAAI,KAAM,UAAU;MACnE,IAAI,YAAY,KAAK,YAAY,GAAG,UAAU;MAC9C,IAAI,YAAY,KAAK,YAAY,GAAG,UAAU;MAC9C,UAAU;CAEf,MAAM,YACJ,oBAAoB,UAAU,QAAQ,CAAC,EAAE,eAAe,UAAU,QAAQ,CAAC,OAC1E,oBAAoB,OAAO,wBAAwB,oBAAoB;CAE1E,OAAO;EAAE;EAAM;EAAW;EAAiB;EAAS;CAAU;AAChE;;AAGA,SAAgB,WAAW,OAAwB,YAAY,IAAoB;CACjF,OAAO,MAAM,OAAO,MAAM,MAAM,EAAE,YAAY,SAAS,CAAC,EAAE,KAAK;AACjE;AAIA,SAAS,gBACP,IACA,WACA,YACA,KAC+B;CAC/B,IAAI,GAAG,SAAS,GAAG,OAAO;EAAE,KAAK,GAAG,MAAM;EAAG,MAAM,GAAG,MAAM;CAAE;CAC9D,MAAM,UAAU,IAAI,MAAc,SAAS;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,KAAK;EAClC,IAAI,MAAM;EACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,OAAO,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;EAC1E,QAAQ,KAAK,MAAM,GAAG;CACxB;CACA,QAAQ,MAAM,GAAG,MAAM,IAAI,CAAC;CAC5B,MAAM,QAAQ,IAAI;CAClB,OAAO;EACL,KAAK,QAAQ,KAAK,MAAO,QAAQ,IAAK,SAAS;EAC/C,MAAM,QAAQ,KAAK,IAAI,YAAY,GAAG,KAAK,MAAM,IAAI,QAAQ,KAAK,SAAS,IAAI,CAAC;CAClF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC9JA,eAAsB,gBAAgB,MAAqD;CACzF,MAAM,SAA8B,CAAC;CACrC,KAAK,MAAM,UAAU,KAAK,SAAS;EACjC,MAAM,IAAI,MAAM,KAAK,YAAY,MAAM;EACvC,OAAO,KAAK;GACV,UAAU,OAAO;GACjB,MAAM,OAAO;GACb,OAAO,EAAE;GACT,SAAS,EAAE;GACX,KAAK,EAAE;GACP,SAAS,EAAE;EACb,CAAC;CACH;CACA,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;CACzD,MAAM,WAAW,OAAO,UAAU,IAAI,YAAY,MAAM,IAAI;CAC5D,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAO,EAAE,QAAQ,EAAE,QAAQ,IAAI,CAAE;CAChE,OAAO;EAAE,aAAa,KAAK;EAAa,QAAQ;EAAQ;EAAU;CAAK;AACzE;;AAqBA,eAAsB,QAAW,MAAkE;CACjG,IAAI,KAAK,KAAK,GAAG,MAAM,IAAI,gBAAgB,wBAAwB;CACnE,MAAM,WAAgB,CAAC;CACvB,MAAM,SAAmB,CAAC;CAC1B,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,GAAG,KAAK;EAC/B,MAAM,IAAI,MAAM,KAAK,OAAO,CAAC;EAC7B,SAAS,KAAK,CAAC;EACf,OAAO,KAAK,MAAM,KAAK,QAAQ,CAAC,CAAC;CACnC;CACA,IAAI,YAAY;CAChB,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,IAAI,OAAO,KAAM,OAAO,YAAa,YAAY;CACzF,MAAM,YAAY,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CAC7D,OAAO;EACL,MAAM,SAAS;EACf,WAAW,OAAO;EAClB;EACA;EACA;CACF;AACF;;;;;AA0BA,eAAsB,gBACpB,MACmC;CACnC,IAAI,KAAK,KAAK,GAAG,MAAM,IAAI,gBAAgB,gCAAgC;CAC3E,MAAM,WAAgB,CAAC;CACvB,MAAM,YAAoC,CAAC;CAC3C,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,GAAG,KAAK;EAC/B,MAAM,IAAI,MAAM,KAAK,OAAO,CAAC;EAC7B,SAAS,KAAK,CAAC;EACf,MAAM,MAAM,KAAK,UAAU,CAAC;EAC5B,UAAU,QAAQ,UAAU,QAAQ,KAAK;CAC3C;CACA,IAAI,SAAS;CACb,IAAI,MAAM;CACV,KAAK,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,SAAS,GAC3C,IAAI,IAAI,KAAK;EACX,MAAM;EACN,SAAS;CACX;CAEF,MAAM,iBAAiB,SAAS,MAAM,MAAM,KAAK,UAAU,CAAC,MAAM,MAAM,KAAK,SAAS;CACtF,OAAO;EACL;EACA,WAAW,MAAM,KAAK;EACtB;EACA;EACA;CACF;AACF;AAeA,SAAgB,eAAe,QAAgD;CAC7E,MAAM,aAAiC,CAAC;CACxC,KAAK,MAAM,KAAK,QAKd,IAAI,CAJc,OAAO,MACtB,MACC,MAAM,KAAK,EAAE,QAAQ,EAAE,QAAQ,EAAE,SAAS,EAAE,UAAU,EAAE,OAAO,EAAE,QAAQ,EAAE,QAAQ,EAAE,MAE5E,GAAG,WAAW,KAAK,CAAC;CAEnC,OAAO,WAAW,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,IAAI;AAClD;AAIA,SAAS,YAAY,QAAqC;CAIxD,MAAM,KAAK,OAAO,KAAK,MAAM,KAAK,IAAI,KAAK,IAAI,OAAO,EAAE,IAAI,CAAC,CAAC;CAC9D,MAAM,KAAK,OAAO,KAAK,MAAM,EAAE,KAAK;CACpC,MAAM,IAAI,GAAG;CACb,MAAM,KAAK,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3C,MAAM,KAAK,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3C,IAAI,MAAM;CACV,IAAI,MAAM;CACV,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,QAAQ,GAAG,KAAM,OAAO,GAAG,KAAM;EACjC,QAAQ,GAAG,KAAM,OAAO;CAC1B;CACA,OAAO,QAAQ,IAAI,IAAI,MAAM;AAC/B;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACpIA,eAAsB,sBACpB,OACA,OAAkC,CAAC,GACA;CACnC,MAAM,MAAM,KAAK,OAAO;CACxB,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,QAAQ,KAAK,cAAc;CAEjC,IAAI,CAAC,MAAM,aAAa,CAAC,MAAM,cAC7B,MAAM,IAAI,gBACR,0EACF;CAEF,MAAM,YACJ,MAAM,aAAc,MAAM,QAAQ,IAAI,MAAM,UAAU,KAAK,MAAM,MAAM,aAAc,MAAM,CAAC,CAAC,CAAC;CAChG,IAAI,UAAU,WAAW,MAAM,UAAU,QACvC,MAAM,IAAI,gBACR,2CAA2C,UAAU,OAAO,eAAe,MAAM,UAAU,QAC7F;CAIF,MAAM,aAAa,MAAM,QAAQ,IAAI,MAAM,UAAU,KAAK,MAAM,MAAM,QAAQ,CAAC,CAAC,CAAC;CACjF,MAAM,aAAa,MAAM,QAAQ,IAAI,UAAU,KAAK,MAAM,MAAM,QAAQ,CAAC,CAAC,CAAC;CAE3E,MAAM,cAAc,MAAM,UAAU,KAAK,GAAG,OAAO;EACjD,YAAY,MAAM,WAAW,CAAC;EAC9B,eAAe,WAAW;EAC1B,gBAAgB,WAAW;EAC3B,OAAO,WAAW,KAAM,WAAW;EACnC,QAAQ;CACV,EAAE;CAGF,MAAM,QAAQ,YAAY,QAAQ,MAAM,EAAE,iBAAiB,SAAS,EAAE,kBAAkB,KAAK;CAC7F,IAAI,MAAM,SAAS,GACjB,OAAO;EACL;EACA,YAAY;GAAE,GAAG;GAAG,GAAG;EAAE;EACzB,aAAa;EACb,WAAW;EACX,wBAAwB;EACxB,QAAQ,mCAAmC,MAAM,OAAO;EACxD,GAAG,MAAM;CACX;CAKF,MAAM,aAAa,mBAFD,MAAM,KAAK,MAAM,EAAE,aAES,GAD5B,MAAM,KAAK,MAAM,EAAE,cACoB,CAAC;CAC1D,MAAM,SAAS,MAAM,KAAK,MAAM,EAAE,KAAK;CACvC,MAAM,eAAe,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CACrD,MAAM,SAAS,aAAa,KAAK,MAAM,aAAa,SAAS,CAAC;CAC9D,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CAOxD,MAAM,EAAE,YAAY,kBADJ,MAAM,KAAK,MAAM,KAAK,IAAI,GAAG,KAAK,IAAI,MAAM,IAAI,KAAK,IAAI,EAAE,KAAK,IAAI,CAAC,CAAC,CAC1C,GAAG,GAAG;CAClD,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;EACrC,MAAM,IAAI,MAAM;EAChB,MAAM,MAAM,YAAY,WAAW,MAAM,EAAE,eAAe,EAAE,UAAU;EACtE,IAAI,OAAO,GAAG,YAAY,IAAI,CAAE,SAAS,QAAQ;CACnD;CAEA,MAAM,yBAAyB,WAAW,IAAI,OAAO,UAAU,CAAC;CAOhE,OAAO;EACL;EACA;EACA,aAAa;EACb,WAAW;EACX;EACA,QAZa,yBACX,YAAY,WAAW,EAAE,QAAQ,CAAC,EAAE,KAAK,IAAI,mBAAmB,OAAO,QAAQ,CAAC,EAAE,KAAK,kBACvF,WAAW,KAAK,MACd,uCAAuC,WAAW,EAAE,QAAQ,CAAC,EAAE,KAC/D,8CAA8C,OAAO,QAAQ,CAAC,EAAE;EASpE,GAAG,MAAM;CACX;AACF;;;;;;;AAUA,SAAgB,gBACd,aACA,UAAiD,GAAG,MAAM,GAAG,EAAE,IAAK,IAAI,KAAM,GAAA,CAAI,SAAS,EAAE,KACpE;CACzB,OAAO;EACL,MAAM;EACN,MAAM,UAAU;GACd,IAAI,SAAS,SAAS;GACtB,YAAY,SAAS,IAAI,MAAM;IAC7B,MAAM,cAAc,OAAO,IAAI,CAAC;IAChC,MAAM,KAAK,IAAI,OAAO,MAAM,YAAY,EAAE,EAAE,MAAM,GAAG;IACrD,SAAS,OAAO,QAAQ,IAAI,WAAW;GACzC,CAAC;GACD,OAAO;IAAE,GAAG;IAAU;GAAO;EAC/B;CACF;AACF;;;;;;AAOA,SAAgB,aACd,gBACA,MACyB;CACzB,MAAM,MAAM,WAAW,IAAI;CAC3B,OAAO;EACL,MAAM;EACN,MAAM,UAAU;GACd,MAAM,YAAY,eAAe,SAAS,QAAQ,GAAG;GACrD,OAAO;IAAE,GAAG;IAAU,QAAQ;GAAU;EAC1C;CACF;AACF;;;;;;AAOA,SAAgB,uBACd,QACA,WAAgC,UACP;CACzB,OAAO;EACL,MAAM;EACN,MAAM,UAAU;GACd,MAAM,SACJ,aAAa,WAAW,GAAG,OAAO,GAAG,SAAS,WAAW,GAAG,SAAS,OAAO,GAAG;GACjF,OAAO;IAAE,GAAG;IAAU;GAAO;EAC/B;CACF;AACF;AAIA,SAAS,YAAY,GAAmB;CACtC,OAAO,EAAE,QAAQ,uBAAuB,MAAM;AAChD;;;;;;;;;;;;;;;AChPA,SAAgB,oBAAoB,MAAwC;CAI1E,iBAAiB,MAAM,kBAAkB;CACzC,MAAM,EAAE,WAAW,KAAK;CACxB,IAAI,WAAW,QAAQ,CAAC,OAAO,SAAS,MAAM,GAAG,OAAO;CACxD,OAAO;AACT;;AAGA,SAAgB,oBAAoB,MAAkC;CACpE,OAAO,KAAK,QAAQ,mBAAmB;AACzC;AAgEA,SAAS,KAAK,OAAyC,KAAa,MAA+B;CACjG,MAAM,WAAW,MAAM,IAAI,GAAG;CAC9B,IAAI,aAAa,KAAA,GAAW,MAAM,IAAI,KAAK,CAAC,IAAI,CAAC;MAC5C,SAAS,KAAK,IAAI;AACzB;AAEA,SAAS,gBAAgB,OAAsD;CAC7E,MAAM,4BAAY,IAAI,IAAiC;CACvD,MAAM,wBAAQ,IAAI,IAAiC;CACnD,KAAK,MAAM,QAAQ,OAAO;EACxB,KAAK,WAAW,KAAK,YAAY,IAAI;EACrC,KAAK,OAAO,KAAK,QAAQ,IAAI;CAC/B;CACA,OAAO;EAAE;EAAW;CAAM;AAC5B;;;;;;;;;;;;;;;;;;;AAoBA,SAAS,kBAAkB,OAAwB,IAAwB;CACzE,MAAM,WAAW,MAAM,UAAU,IAAI,EAAE,KAAK,CAAC;CAC7C,MAAM,OAAO,MAAM,MAAM,IAAI,EAAE,KAAK,CAAC;CACrC,IAAI,SAAS,SAAS,GAAG,OAAO;EAAE,MAAM;EAAa,OAAO,SAAS;CAAO;CAC5E,MAAM,QAAQ,SAAS;CACvB,IAAI,UAAU,KAAA,GAAW;EACvB,IAAI,KAAK,MAAM,SAAS,SAAS,KAAK,GACpC,OAAO;GAAE,MAAM;GAAa,OAAO,IAAI,KAAK,QAAQ,SAAS,SAAS,KAAK,CAAC,CAAC;EAAO;EAEtF,OAAO;GAAE,MAAM;GAAY,MAAM;EAAM;CACzC;CACA,IAAI,KAAK,SAAS,GAAG,OAAO;EAAE,MAAM;EAAa,OAAO,KAAK;CAAO;CACpE,MAAM,OAAO,KAAK;CAClB,OAAO,SAAS,KAAA,IAAY,EAAE,MAAM,UAAU,IAAI;EAAE,MAAM;EAAY,MAAM;CAAK;AACnF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoDA,SAAgB,yBACd,OACA,OACA,SACA,aACA,SACmB;CACnB,IAAI,YAAY,KAAA,KAAa,YAAY,MACvC,MAAM,IAAI,MACR,GAAG,YAAY,SAAS,MAAM,YAAY,YAAY,iBAAiB,YAAY,QAAQ,wDAC7F;CAEF,MAAM,QAAQ,gBAAgB,QAAQ,KAAK;CAC3C,MAAM,QAA2B;EAC/B,UAAU,CAAC;EACX,YAAY;EACZ,gBAAgB;EAChB,WAAW,CAAC;CACd;CACA,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAA6B,CAAC;EACpC,IAAI,YAAY;EAChB,KAAK,MAAM,MAAM,MAAM,IAAI,GAAG;GAC5B,MAAM,aAAa,kBAAkB,OAAO,EAAE;GAC9C,IAAI,WAAW,SAAS,WACtB,MAAM,IAAI,MACR,GAAG,YAAY,SAAS,qCAAqC,GAAG,qDAClE;GAEF,IAAI,WAAW,SAAS,aAAa;IACnC,YAAY;IACZ,IAAI,CAAC,MAAM,UAAU,MAAM,UAAU,MAAM,OAAO,EAAE,GAClD,MAAM,UAAU,KAAK;KAAE;KAAI,aAAa,WAAW;IAAM,CAAC;IAE5D;GACF;GACA,MAAM,KAAK,WAAW,IAAI;EAC5B;EACA,KAAK,MAAM,QAAQ,OAAO;GACxB,iBAAiB,MAAM,YAAY,QAAQ;GAC3C,UAAU,IAAI;EAChB;EACA,IAAI,WAAW;GACb,MAAM;GACN;EACF;EACA,IAAI,MAAM,KAAK,mBAAmB,GAAG;GACnC,MAAM;GACN;EACF;EACA,MAAM,SAAS,KAAK,IAAI;CAC1B;CACA,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,yBACd,OACA,OACA,SACA,aACA,SACK;CACL,MAAM,QAAQ,yBAAyB,OAAO,OAAO,SAAS,aAAa,OAAO;CAClF,IAAI,MAAM,iBAAiB,GAAG;EAC5B,MAAM,QAAQ,MAAM,UAAU,KAAK,MAAM,GAAG,EAAE,GAAG,IAAI,EAAE,YAAY,cAAc,CAAC,CAAC,KAAK,IAAI;EAC5F,QAAQ,KACN,IAAI,YAAY,SAAS,YAAY,MAAM,eAAe,YAAY,MAAM,6OAI9E;CACF;CACA,OAAO,MAAM;AACf;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC3MA,MAAM,0BAAkD;CACtD,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;;;;;;;;;AAiBA,eAAsB,UACpB,SACA,SACA,SACyB;CACzB,MAAM,WAAW,yBACf,UACC,MAAM,CAAC,EAAE,aAAa,EAAE,aAAa,GACtC,SACA,uBACF;CACA,MAAM,MAAsB,CAAC;CAC7B,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,CAAC,cAAc,gBAAgB,QAAQ,YAAY,MAAM,QAAQ,IAAI;GACzE,QAAQ,QAAQ,QAAQ,SAAS,EAAE,WAAW,CAAC;GAC/C,QAAQ,QAAQ,QAAQ,SAAS,EAAE,aAAa,CAAC;GACjD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,WAAW,CAAC;GACnD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,aAAa,CAAC;EACvD,CAAC;EACD,IAAI,iBAAiB,gBACnB,MAAM,IAAI,MACR,0BAA0B,EAAE,YAAY,KAAK,EAAE,cAAc,gCAC/D;EAEF,IAAI,KAAK;GACP,QAAQ;GACR;GACA;GACA,QAAQ,EAAE;GACV,MAAM;IACJ,YAAY,EAAE;IACd,iBAAiB,EAAE;IACnB,mBAAmB,EAAE;IACrB,aAAa,EAAE;IACf,eAAe,EAAE;IACjB,aAAa,EAAE,KAAK;IACpB,eAAe,EAAE,KAAK;GACxB;EACF,CAAC;CACH;CACA,OAAO;AACT;;AAGA,SAAgB,WAAW,MAA8B;CACvD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;;;;;;;;;;;;;;;;;;AAkDA,eAAsB,WACpB,OACA,SAC0B;CAC1B,OAAO,kBAAkB,OAAO,OAAO;AACzC;AAEA,eAAe,kBACb,OACA,SAC0B;CAC1B,MAAM,0BAAU,IAAI,IAAiC;CACrD,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,gBAAgB,MAAM,OAAO,GAAG;EACrC,MAAM,MAAM,QAAQ,IAAI,KAAK,KAAK,WAAW,KAAK,CAAC;EACnD,IAAI,KAAK,IAAI;EACb,QAAQ,IAAI,KAAK,KAAK,aAAa,GAAG;CACxC;CAEA,MAAM,OAAwB,CAAC;CAC/B,KAAK,MAAM,CAAC,YAAY,UAAU,QAAQ,QAAQ,GAAG;EACnD,IAAI,MAAM,WAAW,GAAG;EACxB,MAAM,SAA6D,CAAC;EACpE,KAAK,MAAM,QAAQ,OAAO;GACxB,MAAM,SAAS,oBAAoB,IAAI;GACvC,IAAI,WAAW,MAAM;GACrB,OAAO,KAAK;IAAE;IAAM;GAAO,CAAC;EAC9B;EAGA,IAAI,OAAO,SAAS,GAAG;EACvB,MAAM,UAAU,MAAM,QAAQ,IAC5B,OAAO,KAAK,EAAE,WAAW,QAAQ,QAAQ,QAAQ,SAAS,KAAK,MAAM,CAAC,CAAC,CACzE;EACA,MAAM,SAAS,QAAQ;EACvB,IAAI,QAAQ,MAAM,UAAU,UAAU,MAAM,GAC1C,MAAM,IAAI,MACR,yBAAyB,WAAW,qDACtC;EAEF,MAAM,cAAc,MAAM,QAAQ,IAChC,OAAO,KAAK,EAAE,WAAW,QAAQ,QAAQ,QAAQ,aAAa,KAAK,MAAM,CAAC,CAAC,CAC7E;EACA,MAAM,UAAU,OAAO,KAAK,EAAE,aAAa,MAAM;EACjD,MAAM,SAAS,OAAO,KAAK,EAAE,WAAW,KAAK,MAAM;EACnD,KAAK,KAAK;GACR;GACA;GACA;GACA;GACA,MAAM;IACJ;IACA,GAAG,YAAY;IACf,YAAY,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,QAAQ;GAC3D;EACF,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAgB,YAAY,MAA+B;CACzD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;;;;;;;;;;;;;;;;;;;AAwCA,eAAsB,UACpB,OACA,SACyB;CACzB,OAAO,iBAAiB,OAAO,OAAO;AACxC;AAEA,eAAe,iBACb,OACA,SACyB;CACzB,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,0BAA0B,QAAQ,2BAA2B;CACnE,IAAI,CAAC,OAAO,SAAS,uBAAuB,GAC1C,MAAM,IAAI,MAAM,wCAAwC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,QAAQ,OAAO;EAIxB,iBAAiB,MAAM,YAAY;EACnC,IAAI,oBAAoB,IAAI,GAAG;EAC/B,IAAI,CAAC,gBAAgB,MAAM,OAAO,GAAG;EACrC,MAAM,QAAQ,oBAAoB,IAAI;EACtC,IAAI,UAAU,QAAQ,SAAS,yBAAyB;EACxD,IAAI,CAAC,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,UAAU,MACpF;EAEF,IAAI,CAAC,QAAQ,IAAI,GAAG;EACpB,MAAM,SAAS,QAAQ,WAAW,IAAI;EACtC,MAAM,CAAC,QAAQ,cAAc,MAAM,QAAQ,IAAI,CAC7C,QAAQ,QAAQ,QAAQ,SAAS,KAAK,MAAM,CAAC,GAC7C,QAAQ,QAAQ,QAAQ,aAAa,KAAK,MAAM,CAAC,CACnD,CAAC;EACD,MAAM,WAAqC,CAAC;EAC5C,IAAI,QAAQ,SAAS,KAAK;GAAE,MAAM;GAAU,SAAS;EAAO,CAAC;EAC7D,SAAS,KAAK;GAAE,MAAM;GAAQ,SAAS;EAAO,CAAC;EAC/C,SAAS,KAAK;GAAE,MAAM;GAAa,SAAS;EAAW,CAAC;EACxD,KAAK,KAAK;GACR;GACA,MAAM;IACJ,OAAO,KAAK;IACZ,aAAa,KAAK,gBAAgB;IAClC,YAAY,KAAK,KAAK;IACtB;IACA,OAAO,KAAK,OAAO;GACrB;EACF,CAAC;CACH;CACA,OAAO;AACT;AAEA,SAAgB,WAAW,MAA8B;CACvD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;;;;;;;;;;;;;;;;;;;;AA0DA,eAAsB,UACpB,SACA,SACA,SACyB;CACzB,MAAM,WAAW,gBAAgB,SAAS,OAAO;CACjD,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,SAAS,MAAM,QAAQ,QAAQ,QAAQ,SAAS,EAAE,WAAW,CAAC;EACpE,MAAM,gBAAgB,QAAQ,WAC1B,MAAM,QAAQ,QAAQ,QAAQ,SAAS,EAAE,aAAa,EAAE,eAAe,CAAC,IACxE,CAAC;EACL,MAAM,iBAA2B,CAAC;EAClC,KAAK,MAAM,UAAU,eACnB,eAAe,KAAK,MAAM,QAAQ,QAAQ,QAAQ,WAAW,EAAE,aAAa,MAAM,CAAC,CAAC;EAEtF,MAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,WAAW,EAAE,aAAa,EAAE,YAAY,CAAC;EAC1F,MAAM,eAAe,MAAM,QAAQ,QACjC,QAAQ,WAAW,EAAE,eAAe,EAAE,cAAc,CACtD;EACA,KAAK,KAAK;GACR;GACA;GACA;GACA;GACA;GACA,cAAc,EAAE;GAChB,gBAAgB,EAAE;GAClB,aAAa,EAAE;GACf,MAAM;IACJ,aAAa,EAAE;IACf,eAAe,EAAE;IACjB,iBAAiB,EAAE;GACrB;EACF,CAAC;CACH;CACA,OAAO;AACT;;;;;;;;;AAUA,SAAS,uBAAuB,MAAyB,oBAAmC;CAC1F,MAAM,KAAK,KAAK;CAChB,IAAI,KAAK,WAAW,QAAQ,KAAA,GAC1B,MAAM,IAAI,MACR,uBAAuB,GAAG,kBAAkB,KAAK,WAAW,IAAI,qFAClE;CAEF,IAAI,KAAK,UAAU,KAAA,KAAa,KAAK,MAAM,WAAW,GACpD,MAAM,IAAI,MACR,uBAAuB,GAAG,8EAC5B;CAEF,IAAI,KAAK,QAAQ,cACf,MAAM,IAAI,MACR,uBAAuB,GAAG,sFAC5B;CAEF,IAAI,uBAAuB,KAAA,KAAa,KAAK,MAAM,UAAU,oBAC3D,MAAM,IAAI,MACR,uBAAuB,GAAG,OAAO,KAAK,MAAM,OAAO,4BAA4B,mBAAmB,wGACpG;AAEJ;AAEA,MAAM,0BAAkD;CACtD,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;AASA,SAAS,gBACP,SACA,SACqB;CACrB,OAAO,yBACL,UACC,MAAM,CAAC,EAAE,aAAa,EAAE,aAAa,GACtC,SACA,0BACC,SAAS,uBAAuB,MAAM,QAAQ,kBAAkB,CACnE;AACF;AAEA,SAAgB,WAAW,MAA8B;CACvD,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;AAaA,MAAM,kCAA0D;CAC9D,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;;;;AAYA,SAAgB,mBAAmB,aAA2B,SAAqC;CAOjG,MAAM,OANW,yBACf,cACC,MAAM,CAAC,EAAE,KAAK,GACf,SACA,+BAEwC,CAAC,CAAC,KAAK,OAAO;EACtD,OAAO,EAAE;EACT,QAAQ,EAAE;EACV,WAAW,EAAE;EACb,QAAQ,EAAE;EACV,aAAa,EAAE;EACf,QAAQ,EAAE,UAAU;CACtB,EAAE;CACF,OAAO,KAAK,KAAK,MAAM,KAAK,UAAU,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACnF;AAEA,SAAS,gBACP,MACA,SACS;CACT,IAAI,QAAQ,gBAAgB,KAAA,GAAW,OAAO,QAAQ,YAAY,SAAS,KAAK,KAAK,KAAK;CAC1F,OAAO,gBAAgB,MAAM,OAAO;AACtC;;;ACpgBA,MAAM,qCAAqB,IAAI,IAAa;CAHnB;CAAQ;CAAO;AAGkB,CAAC;AAE3D,SAAgB,uBAAuB,OAAiC;CACtE,IAAI,CAAC,MAAM,QAAQ,KAAK,KAAK,MAAM,WAAW,GAC5C,MAAM,IAAI,MAAM,sEAAsE;CAGxF,MAAM,UAA2B,CAAC;CAClC,MAAM,uBAAO,IAAI,IAAmB;CACpC,KAAK,MAAM,UAAU,OAAO;EAC1B,IAAI,CAAC,mBAAmB,IAAI,MAAM,GAChC,MAAM,IAAI,MACR,sCAAsC,KAAK,UAAU,MAAM,EAAE,0CAC/D;EAEF,MAAM,gBAAgB;EACtB,IAAI,KAAK,IAAI,aAAa,GACxB,MAAM,IAAI,MACR,oCAAoC,KAAK,UAAU,aAAa,EAAE,oCACpE;EAEF,KAAK,IAAI,aAAa;EACtB,QAAQ,KAAK,aAAa;CAC5B;CACA,OAAO;AACT;AAyEA,SAAS,SAAS,IAAgD;CAChE,OAAO,CAAC,GAAG,IAAI,IAAI,GAAG,QAAQ,MAAmB,OAAO,MAAM,YAAY,EAAE,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK;AACjG;AAEA,SAAS,mBAAmB,QAA+B;CACzD,IAAI,OAAO,WAAW,GACpB,OAAO;EAAE,GAAG;EAAG,MAAM;EAAM,QAAQ;EAAM,KAAK;EAAM,KAAK;EAAM,KAAK;CAAK;CAE3E,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,IAAI,OAAO;CACjB,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACjD,MAAM,MAAM,KAAK,MAAM,IAAI,CAAC;CAC5B,MAAM,SAAS,IAAI,MAAM,KAAK,OAAO,MAAM,KAAM,OAAO,QAAS,IAAI,OAAO;CAC5E,MAAM,WAAW,OAAO,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,IAAI;CACnE,OAAO;EAAE;EAAG;EAAM;EAAQ,KAAK,OAAO;EAAK,KAAK,OAAO,IAAI;EAAK,KAAK,KAAK,KAAK,QAAQ;CAAE;AAC3F;AAEA,SAAS,sBAAsB,OAA4C;CACzE,MAAM,SAAuC;EAAE,QAAQ;EAAG,KAAK;EAAG,SAAS;EAAG,QAAQ;CAAE;CACxF,IAAI,QAAQ;CACZ,IAAI,SAAS;CACb,IAAI,OAAO;CACX,IAAI,sBAAsB;CAC1B,MAAM,UAAoB,CAAC;CAC3B,KAAK,MAAM,QAAQ,OAAO;EACxB,OAAO,KAAK,KAAK,UAAU;EAC3B,SAAS,KAAK,KAAK,aAAa;EAChC,UAAU,KAAK,KAAK,cAAc;EAClC,IAAI,KAAK,KAAK,QAAQ,MAAM;OACvB,QAAQ,KAAK,KAAK;EACvB,MAAM,KAAK,oBAAoB,IAAI;EACnC,IAAI,OAAO,MAAM,QAAQ,KAAK,EAAE;CAClC;CACA,OAAO;EACL,SAAS,MAAM;EACf,eAAe,QAAQ;EACvB;EACA,QAAQ,mBAAmB,OAAO;EAClC,QAAQ,SAAS,MAAM,KAAK,MAAM,EAAE,OAAO,KAAK,CAAC;EACjD,cAAc,SAAS,MAAM,KAAK,MAAM,EAAE,OAAO,WAAW,CAAC;EAC7D,YAAY,SAAS,MAAM,KAAK,MAAM,EAAE,OAAO,cAAc,CAAC;EAC9D,aAAa;GAAE,OAAO;GAAO,QAAQ;EAAO;EAC5C,cAAc;EACd;CACF;AACF;;;;;;;;AASA,eAAsB,eACpB,OACA,SACA,QACA,aAC0B;CAC1B,IAAI,MAAM,WAAW,GACnB,MAAM,IAAI,MAAM,yEAAyE;CAE3F,MAAM,UAAU,uBAAuB,OAAO,YAAY,KAAA,IAAY,CAAC,KAAK,IAAI,OAAO,OAAO;CAC9F,MAAM,QAAgC,CAAC;CACvC,MAAM,YAAoD,CAAC;CAE3D,IAAI,QAAQ,SAAS,MAAM,GAAG;EAC5B,MAAM,OAAO,MAAM,WAAW,OAAO,OAAO;EAC5C,YAAY,QAAQ,KAAK,MAAM;EAC/B,MAAM,sBAAsB,YAAY,IAAI;EAC5C,UAAU,OAAO,KAAK;CACxB;CACA,IAAI,QAAQ,SAAS,KAAK,GAAG;EAC3B,MAAM,OAAO,MAAM,UAAU,OAAO,OAAO;EAC3C,YAAY,OAAO,KAAK,MAAM;EAC9B,MAAM,qBAAqB,WAAW,IAAI;EAC1C,UAAU,MAAM,KAAK;CACvB;CACA,IAAI,QAAQ,SAAS,KAAK,GAAG;EAC3B,IAAI,CAAC,aACH,MAAM,IAAI,MAAM,yEAAyE;EAE3F,MAAM,OAAO,MAAM,UAAU,YAAY,SAAS,YAAY,SAAS,EAAE,MAAM,CAAC;EAChF,YAAY,OAAO,KAAK,MAAM;EAC9B,MAAM,qBAAqB,WAAW,IAAI;EAC1C,UAAU,MAAM,KAAK;CACvB;CACA,IAAI,CAAC,OAAO,KAAK,KAAK,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,QAAQ,KAAK,KAAK,SAAS,QAAQ,CAAC,GACzF,MAAM,IAAI,MAAM,6CAA6C;CAG/D,MAAM,WAA8B;EAClC,GAAG;EACH;EACA;EACA,OAAO,sBAAsB,KAAK;CACpC;CACA,MAAM,mBAAmB,GAAG,KAAK,UAAU,UAAU,MAAM,CAAC,EAAE;CAC9D,MAAM,kBAAkB,oBAAoB,QAAQ;CACpD,OAAO;EAAE;EAAU;CAAM;AAC3B;AAEA,SAAS,YAAY,QAAuB,MAAoB;CAC9D,IAAI,SAAS,GACX,MAAM,IAAI,MAAM,8BAA8B,OAAO,oCAAoC;AAE7F;AAEA,SAAS,IAAI,GAAmB;CAC9B,OAAO,IAAI,IAAI,IAAA,CAAK,QAAQ,CAAC,EAAE;AACjC;AAEA,SAAS,KAAK,OAA8B;CAC1C,OAAO,UAAU,OAAO,QAAQ,MAAM,QAAQ,CAAC;AACjD;;AAGA,SAAgB,oBAAoB,GAA8B;CAChE,MAAM,IAAI,EAAE;CACZ,MAAM,QAAQ,EAAE,WAAW;CAC3B,MAAM,aAAc;EAAC;EAAU;EAAO;EAAW;CAAQ,CAAC,CACvD,KAAK,MAAM,SAAS,EAAE,MAAM,EAAE,OAAO,GAAG,IAAI,IAAI,EAAE,OAAO,KAAK,KAAK,EAAE,EAAE,CAAC,CACxE,KAAK,IAAI;CACZ,MAAM,WACJ,EAAE,sBAAsB,IACpB,aAAa,EAAE,oBAAoB,sCACnC;CACN,MAAM,gBAAgB,EAAE,OAAO,SAAS;CACxC,OAAO;EACL,cAAc,EAAE,KAAK,MAAM,EAAE,QAAQ;EACrC;EACA,eAAe,EAAE,OAAO,kBAAkB,EAAE,aAAa,kBAAkB,EAAE;EAC7E;EACA;EACA,eAAe,EAAE,OAAO,OAAO,gBAAgB,kCAAkC;EACjF,iBAAiB,EAAE,OAAO;EAC1B,sBAAsB,EAAE,OAAO;EAC/B;EACA;EACA,iCAAiC,EAAE;EACnC,yBAAyB,EAAE;EAC3B,kBAAkB,EAAE,QAAQ,KAAK,MAAM,GAAG,EAAE,IAAI,EAAE,UAAU,MAAM,EAAE,OAAO,CAAC,CAAC,KAAK,IAAI;EACtF;EACA;EACA;EACA;EACA,OAAO,EAAE,OAAO,EAAE,UAAU,KAAK,EAAE,OAAO,IAAI,EAAE,YAAY,KAAK,EAAE,OAAO,MAAM,EAAE,SAAS,KAAK,EAAE,OAAO,GAAG,EAAE,SAAS,KAAK,EAAE,OAAO,GAAG,EAAE,SAAS,KAAK,EAAE,OAAO,GAAG;EACpK;EACA;EACA,iBAAiB,EAAE,OAAO,KAAK,IAAI;EACnC,yCAAyC,EAAE,aAAa,OAAO;EAC/D,kBAAkB,EAAE,WAAW,KAAK,IAAI;EACxC,iBAAiB,EAAE,YAAY,MAAM,QAAQ,EAAE,YAAY,OAAO,oBAAoB,EAAE,aAAa,QAAQ,CAAC,IAAI;EAClH;EACA;EACA,0BAA0B,EAAE,cAAc,sBAAsB;EAChE,YAAY,EAAE,cAAc,QAAQ,QAAQ,KAAK,+BAA+B,EAAE,cAAc,yBAAyB,QAAQ;EACjI;EACA;EACA,EAAE;EACF;EACA;EACA,EAAE;EACF;EACA;EACA,EAAE;EACF;EACA;EACA;EACA;CACF,CAAC,CAAC,KAAK,IAAI;AACb;;;;;;;;;;;;;;;;;;;;;;;;;AC/QA,SAAgB,eAAe,SAAyB,YAAwC;CAC9F,UAAU,QAAQ,UAAU,GAAG,EAAE,WAAW,KAAK,CAAC;CAClD,MAAM,WAAW,WAAW,UAAU,IAAI,WAAW,UAAU,IAAI,CAAC;CACpE,MAAM,OAAO,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACjD,MAAM,QAAkB,CAAC;CACzB,IAAI,WAAW;CACf,IAAI,UAAU;CACd,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,KAAK,IAAI,EAAE,KAAK,GAAG;GACrB;GACA;EACF;EACA,KAAK,IAAI,EAAE,KAAK;EAChB,MAAM,KAAK,KAAK,UAAU,CAAC,CAAC;EAC5B;CACF;CACA,IAAI,MAAM,SAAS,GAAG,eAAe,YAAY,GAAG,MAAM,KAAK,IAAI,EAAE,GAAG;CACxE,OAAO;EAAE;EAAU;EAAS,OAAO,SAAS,SAAS;CAAS;AAChE;;AAGA,SAAgB,WAAW,YAAoC;CAC7D,IAAI,CAAC,WAAW,UAAU,GAAG,OAAO,CAAC;CACrC,MAAM,MAAsB,CAAC;CAC7B,KAAK,MAAM,QAAQ,aAAa,YAAY,MAAM,CAAC,CAAC,MAAM,IAAI,GAC5D,IAAI,KAAK,KAAK,GAAG,IAAI,KAAK,KAAK,MAAM,IAAI,CAAiB;CAE5D,OAAO;AACT;;;;;;AAOA,SAAS,SAAS,GAAgC;CAChD,MAAM,IAAI,cAAc,CAAC;CACzB,OAAO,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,IAAI,IAAI;AAC3D;;;;;;;;;;;;;;AAwBA,eAAsB,uBACpB,YACA,QACA,OAAuB,CAAC,GACE;CAC1B,IAAI,UAAU,WAAW,UAAU,CAAC,CAAC,QAClC,MAAM,OAAO,EAAE,WAAW,YAAY,OAAO,EAAE,eAAe,QACjE;CACA,IAAI,KAAK,QAAQ,UAAU,QAAQ,QAAQ,MAAM,KAAK,OAAQ,SAAS,EAAE,QAAQ,CAAC;CAClF,UAAU,QAAQ,QAAQ,MAAM,SAAS,CAAC,MAAM,IAAI;CACpD,IAAI,KAAK,YAAY,MACnB,UAAU,QAAQ,QAAQ,MAAM;EAC9B,MAAM,SAAS,SAAS,CAAC;EACzB,OAAO,WAAW,QAAQ,UAAU,KAAK;CAC3C,CAAC;CAGH,MAAM,OAAO,IAAI,IACf,QAAQ,KAAK,MAAM,CAAC,EAAE,OAAO;EAAE,QAAQ,EAAE;EAAS,YAAY,EAAE;CAAY,CAAC,CAAC,CAChF;CACA,MAAM,UAAU;EACd,WAAW,OAAe,KAAK,IAAI,EAAE,CAAC,EAAE,UAAU;EAClD,eAAe,OAAe,KAAK,IAAI,EAAE,CAAC,EAAE,cAAc;EAC1D,0BAA0B,KAAK;CACjC;CACA,MAAM,EAAE,SAAS,MAAM,gBAAgB,SAAS,IAAI,mBAAmB,CAAC;CACxE,OAAO,eAAe,MAAM,SAAS,MAAM;AAC7C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACjBA,SAAgB,4BACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAElD,IAAI,aAAa,WAAW,GAC1B,OAAO,aAAa;CAGtB,MAAM,UAAoB,CAAC;CAC3B,MAAM,kBAA4B,CAAC;CACnC,IAAI,OAAO;CACX,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,gEAAgE,EAAE,MAAM,EAC1E;EAEF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,MAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;EAC7C,QAAQ,KAAK,CAAC;EACd,gBAAgB,KAAK,IAAI,CAAC;EAC1B,IAAI,IAAI,MAAM,OAAO;CACvB;CACA,MAAM,IAAI,QAAQ;CAClB,MAAM,QAAQ,gBAAgB,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3D,MAAM,WAAW,gBAAgB,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;CAC9F,MAAM,OAAO,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC;CAC9C,MAAM,QAAQ,QAAQ,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;CACnD,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAE9C,OAAO;EACL;EACA,eAAe,KAAK,KAAK,WAAW,CAAC;EACrC,qBAAqB;EACrB;EACA,qBAAqB;CACvB;AACF;;;;;;AAOA,SAAgB,kCACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAClD,IAAI,aAAa,WAAW,GAAG,OAAO,aAAa;CAEnD,MAAM,UAAoB,CAAC;CAC3B,MAAM,UAAoB,CAAC;CAC3B,IAAI,OAAO;CACX,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,sEAAsE,EAAE,MAAM,EAChF;EAEF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,QAAQ,KAAK,CAAC;EACd,QAAQ,KAAK,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI,CAAC;EACjD,IAAI,IAAI,MAAM,OAAO;CACvB;CACA,MAAM,OAAO,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC;CAC9C,MAAM,QAAQ,QAAQ,QAAQ,GAAG,GAAG,MAAM,IAAI,IAAI,QAAQ,IAAK,CAAC;CAChE,MAAM,QAAQ,SAAS,IAAI,IAAI,QAAQ;CACvC,MAAM,QAAQ,QAAQ,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;CACnD,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAG9C,MAAM,WADM,QAAQ,KAAK,GAAG,MAAM,KAAK,QAAQ,KAAM,MAClC,CAAC,CAAC,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,OAAO,IAAI;CAC7E,OAAO;EACL;EACA,eAAe,KAAK,KAAK,QAAQ;EACjC,qBAAqB;EACrB,GAAG,aAAa;EAChB,qBAAqB;CACvB;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,SAAgB,aACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAClD,IAAI,aAAa,WAAW,GAC1B,OAAO;EACL,GAAG,aAAa;EAChB,oBAAoB;GAAE,IAAI;GAAG,aAAa;EAAE;CAC9C;CAGF,MAAM,gBAA0B,CAAC;CACjC,MAAM,qBAAkD;EACtD,IAAI;EACJ,aAAa;CACf;CACA,IAAI,OAAO;CACX,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBAAgB,iDAAiD,EAAE,MAAM,EAAE;EAEvF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,MAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;EAC7C,MAAM,gBAAgB,EAAE;EACxB,MAAM,gBAAgB,EAAE;EACxB,MAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB,KAAA;EAClE,MAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB,KAAA;EAClE,IAAI,kBAAkB,eACpB,MAAM,IAAI,gBACR,4EAA4E,EAAE,MAAM,EACtF;EAGF,IAAI,iBAAiB,eAAe;GAClC,IAAI,CAAC,OAAO,SAAS,aAAa,KAAK,CAAC,OAAO,SAAS,aAAa,GACnE,MAAM,IAAI,gBACR,iEAAiE,EAAE,MAAM,EAC3E;GAEF,MAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;GAC3D,MAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;GAC3D,cAAc,KAAK,aAAa,KAAK,IAAI,WAAW;GACpD,mBAAmB,MAAM;EAC3B,OAAO;GACL,cAAc,KAAK,IAAI,CAAC;GACxB,mBAAmB,eAAe;EACpC;EACA,IAAI,IAAI,MAAM,OAAO;EACrB,QAAQ;EACR,SAAS,IAAI;CACf;CACA,MAAM,IAAI,cAAc;CACxB,MAAM,QAAQ,cAAc,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACzD,MAAM,WAAW,cAAc,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;CAC5F,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAC9C,OAAO;EACL;EACA,eAAe,KAAK,KAAK,WAAW,CAAC;EACrC,qBAAqB;EACrB;EACA,qBAAqB;EACrB;CACF;AACF;;;;;;AAOA,SAAgB,qBACd,cACA,OAAyB,CAAC,GACmD;CAC7E,OAAO;EACL,KAAK,4BAA4B,cAAc,IAAI;EACnD,OAAO,kCAAkC,cAAc,IAAI;EAC3D,IAAI,aAAa,cAAc,IAAI;CACrC;AACF;AAIA,SAAS,eAAkC;CACzC,OAAO;EAAE,OAAO;EAAG,eAAe;EAAG,qBAAqB;EAAG,GAAG;EAAG,qBAAqB;CAAE;AAC5F;AAEA,SAAS,MAAM,GAAW,IAAY,IAAoB;CACxD,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,CAAC,CAAC;AACrC;;;;;;;AChQA,IAAa,+BAAb,MAAgE;CAC9D;CACA,aAA4D;CAE5D,YAAY,MAA2C;EACrD,KAAK,OAAO;CACd;CAEA,MAAM,gBAAgB,MAA2C;EAC/D,MAAM,YAAY,KAAK,KAAK,oBAAoB;EAChD,MAAM,WAA0B,CAAC;EAIjC,MAAM,cAAc,KAAK,QAAQ,MAAM;GACrC,MAAM,QAAQ,aAAa,CAAC;GAC5B,OAAO,OAAO,UAAU,YAAY,QAAQ;EAC9C,CAAC;EACD,IAAI,YAAY,WAAW,GAAG,OAAO;EAIrC,MAAM,0BAAU,IAAI,IAAyB;EAC7C,KAAK,MAAM,KAAK,aAAa;GAC3B,MAAM,MAAM,QAAQ,IAAI,EAAE,WAAW,KAAK,CAAC;GAC3C,IAAI,KAAK,CAAC;GACV,QAAQ,IAAI,EAAE,aAAa,GAAG;EAChC;EAEA,KAAK,MAAM,CAAC,aAAa,UAAU,QAAQ,QAAQ,GAAG;GACpD,MAAM,YACJ,MAAM,QAAQ,GAAG,MAAM;IACrB,MAAM,QAAQ,aAAa,CAAC;IAC5B,IAAI,UAAU,KAAA,GACZ,MAAM,IAAI,MAAM,eAAe,EAAE,MAAM,gCAAgC;IAEzE,OAAO,IAAI;GACb,GAAG,CAAC,IAAI,MAAM;GAChB,SAAS,KAAK;IACZ,MAAM,aAAa;IACnB,aAAa,GAAG,YAAY,YAAY,UAAU,MAAM,MAAM,OAAO,gBAAgB,UAAU,QAAQ,CAAC,EAAE;IAC1G,UAAU;KACR,QAAQ,MAAM,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,MAAM,EAAE,KAAK;KAC5C,SAAS,MAAM;IACjB;GACF,CAAC;EACH;EACA,OAAO;CACT;CAEA,MAAM,cAAc,UAAoD;EACtE,IAAI,SAAS,WAAW,GAAG,OAAO,CAAC;EAInC,IAAI,KAAK,eAAe,MACtB,OAAO,CACL;GACE,MAAM;GACN,SAAS,EAAE,WAAW,mCAAmC;GACzD,WACE;EACJ,CACF;EAGF,MAAM,sBAAsB,KAAK,KAAK,uBAAuB;EAC7D,MAAM,UAA4B,CAAC;EAEnC,KAAK,MAAM,WAAW,KAAK,WAAW,QAAQ;GAC5C,IAAI,QAAQ,YAAY,gBAAgB;GACxC,IAAI,KAAK,IAAI,QAAQ,QAAQ,KAAK,qBAAqB;GACvD,QAAQ,KAAK;IACX,MAAM;IACN,SAAS;KACP,QAAQ,QAAQ;KAChB,QAAQ;KACR,UAAU,QAAQ;KAClB,aAAa,QAAQ;IACvB;IACA,WAAW,gCAAgC,QAAQ,SAAS,QAAQ,CAAC,EAAE,MAAM,QAAQ,YAAY;IACjG,eAAe,CAAC,KAAK,IAAI,GAAG,MAAO,KAAK,IAAI,QAAQ,QAAQ,CAAC;GAC/D,CAAC;EACH;EACA,KAAK,MAAM,WAAW,KAAK,WAAW,OAAO,MAAM,GAAG,CAAC,GAAG;GACxD,IAAI,QAAQ,YAAY,gBAAgB;GACxC,QAAQ,KAAK;IACX,MAAM;IACN,SAAS;KACP,QAAQ,QAAQ;KAChB,QAAQ;KACR,UAAU,QAAQ;KAClB,aAAa,QAAQ;IACvB;IACA,WAAW,gCAAgC,QAAQ,SAAS,QAAQ,CAAC,EAAE,MAAM,QAAQ,YAAY;IACjG,eAAe,KAAK,IAAI,GAAG,KAAK,IAAI,QAAQ,QAAQ,IAAI,EAAG,IAAI;GACjE,CAAC;EACH;EACA,OAAO;CACT;CAEA,MAAM,YAAY,SAA2B,UAAmD;EAG9F,OAAO;GACL,GAAG;GACH,SAAS,CAAC,GAAG,SAAS,SAAS,GAAG,OAAO;EAC3C;CACF;CAEA,MAAM,eAAe,MAAiD;EAqCpE,OAAO;GACL;GACA,MAAM,CAAC;GACP,cAAc;IAlCd,SAAS;IACT,aAAa,KAAK;IAClB,YAAY,KAAK;IACjB,UAAU;KACR,gBAAgB;KAChB,uBAAuB;KACvB,sBAAsB;KACtB,mBAAmB;KACnB,gBAAgB;KAChB,eAAe;KACf,UAAU;KACV,cAAc;KACd,SAAS;KACT,aAAa;KACb,aAAa;KACb,aAAa;KACb,cAAc;KACd,YAAY;KACZ,oBAAoB;KACpB,qBAAqB;KACrB,oBAAoB;KACpB,mBAAmB;KAGnB,iBAAiB,cAAc;KAC/B,gBAAgB,cAAc;IAChC;IACA,QACE;IACF,eAAe;GAKO;EACxB;CACF;;;;;;CAOA,MAAM,iBAAiB,MAA4D;EACjF,MAAM,SAAS,MAAM,yBAAyB;GAC5C;GACA,UAAU,KAAK,KAAK;GACpB,gBAAgB,KAAK,KAAK;GAC1B,SAAS,KAAK,KAAK;EACrB,CAAC;EACD,IAAI,KAAK,KAAK,UAAU,MAAM,KAAK,KAAK,SAAS,MAAM;EACvD,KAAK,aAAa;EAClB,OAAO;CACT;;;;;;CAOA,UAAU,QAA8C;EACtD,KAAK,aAAa;CACpB;CAEA,gBAAuD;EACrD,OAAO,KAAK;CACd;AACF;;AAGA,SAAS,gBAA+B;CACtC,OAAO;EAAE,OAAO;EAAG,UAAU;EAAG,eAAe;EAAG,eAAe;EAAG,cAAc;EAAG,UAAU;CAAE;AACnG;;;;ACrHA,MAAM,gBAA8B;;;;;;;;;;;;;;;;;;;AAmCpC,SAAgB,mBACd,OACA,OAAkC,CAAC,GACP;CAC5B,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,iBAAiB,KAAK;CAC5B,IAAI,mBAAmB,aAAa,KAAK,6BAA6B,MACpE,MAAM,IAAI,MAAM,+EAA6E;CAE/F,IAAI,mBAAmB,SAAS,mBAAmB,UACjD,MAAM,IAAI,MACR,8BAA8B,eAAe,0CAC/C;CAEF,MAAM,aAAa,oBAAoB,OAAO,IAAI;CAElD,OAAO;EAAE,GADM,eAAe,WAAW,MAAM,UAAU,SACxC;EAAG,yBAAyB,WAAW;CAAmB;AAC7E;AAOA,SAAS,oBACP,OACA,MACiB;CACjB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,OAA2B,CAAC;CAClC,IAAI,qBAAqB;CACzB,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,KAAK,UAAU,OAAO;EAC/B,IAAI,CAAC,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,gBAAgB,KAAK,QAAQ,UAAU,MACpF;EAEF,MAAM,QAAQ,oBAAoB,IAAI;EACtC,IAAI,UAAU,MAAM;EACpB,MAAM,cAAc,KAAK;EACzB,IAAI,gBAAgB,QAAQ,gBAAgB,KAAA,KAAa,YAAY,WAAW,GAAG;GACjF;GACA;EACF;EACA,KAAK,KAAK;GACR,YAAY,KAAK,KAAK;GACtB,OAAO,KAAK;GACZ;GACA,MAAM,KAAK,KAAK;GAChB;GAIA,YAAY,KAAK,OAAO,eAAe;GACvC,YAAY,KAAK,OAAO,eAAe;GACvC,OAAO,KAAK,OAAO,SAAS;EAC9B,CAAC;CACH;CACA,OAAO;EAAE;EAAM;CAAmB;AACpC;AAEA,SAAS,eACP,eACA,UACA,WAC6D;CAC7D,MAAM,QAA4B,CAAC;CACnC,IAAI,mBAAmB;CACvB,IAAI,iBAAiB;CACrB,IAAI,iBAAiB;CAErB,IAAI,aAAa,+BAA+B;EAE9C,MAAM,yBAAS,IAAI,IAAgC;EACnD,KAAK,MAAM,KAAK,eAAe;GAC7B,MAAM,MAAM,GAAG,EAAE,WAAW,IAAI,EAAE;GAClC,MAAM,MAAM,OAAO,IAAI,GAAG,KAAK,CAAC;GAChC,IAAI,KAAK,CAAC;GACV,OAAO,IAAI,KAAK,GAAG;EACrB;EAEA,KAAK,MAAM,WAAW,OAAO,OAAO,GAAG;GACrC;GACA,IAAI,QAAQ,SAAS,GAAG;IACtB;IACA;GACF;GACA,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAClC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAAK;IAC3C,MAAM,IAAI,QAAQ;IAClB,MAAM,IAAI,QAAQ;IAClB,IAAI,EAAE,gBAAgB,EAAE,aAAa;IACrC,MAAM,SAAS,SAAS,GAAG,GAAG,EAAE,YAAY,SAAS;IACrD,IAAI,OAAO,SAAS,SAAS,MAAM,KAAK,OAAO,IAAI;SAC9C;GACP;EAEJ;CACF,OAAO,IAAI,aAAa,sBAAsB;EAE5C,MAAM,oCAAoB,IAAI,IAG5B;EACF,KAAK,MAAM,KAAK,eAAe;GAC7B,IAAI,cAAc,kBAAkB,IAAI,EAAE,UAAU;GACpD,IAAI,CAAC,aAAa;IAChB,8BAAc,IAAI,IAAI;IACtB,kBAAkB,IAAI,EAAE,YAAY,WAAW;GACjD;GACA,MAAM,MAAM,YAAY,IAAI,EAAE,WAAW;GACzC,IAAI,KAAK;IACP,IAAI,OAAO,EAAE;IACb,IAAI;GACN,OAAO,YAAY,IAAI,EAAE,aAAa;IAAE,OAAO;IAAG,KAAK,EAAE;IAAO,GAAG;GAAE,CAAC;EACxE;EACA,KAAK,MAAM,CAAC,KAAK,eAAe,kBAAkB,QAAQ,GAAG;GAC3D;GACA,MAAM,MAAM,CAAC,GAAG,WAAW,OAAO,CAAC,CAAC,CAAC,KAAK,SAAS;IACjD,GAAG,IAAI;IACP,OAAO,IAAI,MAAM,IAAI;GACvB,EAAE;GACF,IAAI,IAAI,SAAS,GAAG;IAClB;IACA;GACF;GACA,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAC9B,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;IACvC,MAAM,SAAS,SAAS,IAAI,IAAK,IAAI,IAAK,KAAK,SAAS;IACxD,IAAI,OAAO,SAAS,SAAS,MAAM,KAAK,OAAO,IAAI;SAC9C;GACP;EAEJ;CACF,OAAO;EAEL,MAAM,6BAAa,IAAI,IAAgC;EACvD,KAAK,MAAM,KAAK,eAAe;GAC7B,MAAM,MAAM,WAAW,IAAI,EAAE,UAAU,KAAK,CAAC;GAC7C,IAAI,KAAK,CAAC;GACV,WAAW,IAAI,EAAE,YAAY,GAAG;EAClC;EACA,KAAK,MAAM,CAAC,KAAK,QAAQ,WAAW,QAAQ,GAAG;GAC7C;GACA,IAAI,IAAI,SAAS,GAAG;IAClB;IACA;GACF;GACA,MAAM,SAAS,CAAC,GAAG,GAAG,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;GACxD,MAAM,MAAM,OAAO,OAAO,SAAS;GACnC,MAAM,MAAM,OAAO;GACnB,IAAI,IAAI,gBAAgB,IAAI,aAAa;IACvC;IACA;GACF;GACA,MAAM,SAAS,SAAS,KAAK,KAAK,KAAK,SAAS;GAChD,IAAI,OAAO,SAAS,SAAS,MAAM,KAAK,OAAO,IAAI;QAC9C;EACP;CACF;CAEA,OAAO;EAAE;EAAO;EAAgB;EAAkB;EAAgB;CAAS;AAC7E;AAEA,MAAM,sBAAsB,MAA2C,CACrE,EAAE,aACF,EAAE,aACJ;AAEA,MAAM,0BAAkD;CACtD,UAAU;CACV,aAAa;CACb,SACE;AACJ;AAEA,MAAM,gCAAwD;CAC5D,UAAU;CACV,aAAa;CACb,SACE;AACJ;;;;;;;;;;;;;;;;;;AAmBA,eAAsB,YACpB,SACA,SACA,SACsE;CACtE,MAAM,WAAW,yBACf,SACA,oBACA,SACA,uBACF;CACA,MAAM,MAAmE,CAAC;CAC1E,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,CAAC,cAAc,gBAAgB,QAAQ,YAAY,MAAM,QAAQ,IAAI;GACzE,QAAQ,QAAQ,QAAQ,SAAS,EAAE,WAAW,CAAC;GAC/C,QAAQ,QAAQ,QAAQ,SAAS,EAAE,aAAa,CAAC;GACjD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,WAAW,CAAC;GACnD,QAAQ,QAAQ,QAAQ,aAAa,EAAE,aAAa,CAAC;EACvD,CAAC;EACD,IAAI,iBAAiB,gBACnB,MAAM,IAAI,MACR,4BAA4B,EAAE,YAAY,KAAK,EAAE,cAAc,gCACjE;EAEF,IAAI,KAAK;GAAE,QAAQ;GAAc;GAAQ;EAAS,CAAC;CACrD;CACA,OAAO;AACT;;;;;;;;;;AAWA,SAAgB,kBACd,SACA,SAC2F;CAC3F,OAAO,yBACL,SACA,oBACA,SACA,6BACF,CAAC,CAAC,KAAK,OAAO;EACZ,YAAY,EAAE;EACd,aAAa,EAAE;EACf,eAAe,EAAE;EACjB,QAAQ,EAAE;CACZ,EAAE;AACJ;AAIA,SAAS,SACP,GACA,GACA,YACA,WACgE;CAEhE,IADe,KAAK,IAAI,EAAE,QAAQ,EAAE,KAC3B,IAAI,WAAW,OAAO,EAAE,MAAM,SAAS;CAChD,MAAM,CAAC,QAAQ,YAAY,EAAE,QAAQ,EAAE,QAAQ,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC;CAC7D,MAAM,OAAO,OAAO,SAAS,QAAQ,OAAO,SAAS,SAAS,OAAO,OAAO,OAAO,KAAA;CACnF,OAAO;EACL,MAAM;EACN,MAAM;GACJ;GACA,aAAa,OAAO;GACpB,eAAe,SAAS;GACxB,iBAAiB,OAAO;GACxB,mBAAmB,SAAS;GAC5B,aAAa,OAAO,QAAQ,SAAS;GACrC,QAAQ;IAAE,QAAQ,OAAO;IAAO,UAAU,SAAS;GAAM;GACzD;GACA,MAAM;IACJ,kBAAkB,OAAO;IACzB,oBAAoB,SAAS;IAC7B,kBAAkB,OAAO;IACzB,oBAAoB,SAAS;IAC7B,aAAa,OAAO;IACpB,eAAe,SAAS;GAC1B;EACF;CACF;AACF;;;ACtXA,eAAsB,mBACpB,OACA,OACA,MACuB;CAEvB,MAAM,UAAU,CAAC,GAAG,MADA,MAAM,MAAM,EAAE,MAAM,CAAC,CAChB,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACnE,MAAM,MAAoB,CAAC;CAC3B,IAAI,MAAM;CACV,KAAK,MAAM,QAAQ,SAAS;EAC1B,IAAI,KAAK,aAAa,CAAC,KAAK,UAAU,IAAI,GAAG;EAC7C,IAAI,SAAmD;EACvD,KAAK,MAAM,KAAK,KAAK,SAAS;GAC5B,IAAI,CAAC,EAAE,UAAU,SAAS,KAAK,IAAI,GAAG;GACtC,MAAM,IAAI,MAAM,EAAE,MAAM,IAAI;GAC5B,IAAI,GAAG;IACL,SAAS;IACT;GACF;EACF;EACA,IAAI,CAAC,QAAQ;EACb,IAAI,KAAK;GACP,QAAQ,KAAK;GACb;GACA,WAAW;GACX,MAAM,KAAK;GACX,MAAM,KAAK;GACX,QAAQ,OAAO;GACf,aAAa,OAAO;GACpB,WAAW,OAAO;GAClB,QAAQ,OAAO;EACjB,CAAC;CACH;CACA,OAAO;AACT;AAeA,SAAgB,yBAAyB,aAA+C;CACtF,IAAI,YAAY,WAAW,GACzB,OAAO;EACL,OAAO;EACP,YAAY;EACZ,YAAY;EACZ,mBAAmB;EACnB,iBAAiB;EACjB,gBAAgB;EAChB,gBAAgB;CAClB;CAEF,MAAM,QAAQ,YAAY,EAAE,CAAE;CAC9B,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,IAAI,WAAW;CACf,IAAI,aAAa;CACjB,IAAI,WAA0B;CAC9B,IAAI,OAAO,YAAY,EAAE,CAAE;CAC3B,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,QAAQ,KAAK;EAC3C,MAAM,IAAI,YAAY;EACtB,MAAM,IAAI,EAAE,UAAU;EACtB,QAAQ;EACR,SAAS,IAAI,EAAE;EACf,IAAI,EAAE,SAAS,IAAK;EACpB,IAAI,IAAI,GAAG;GACT,MAAM,QAAQ,EAAE,SAAS;GACzB,IAAI,QAAQ,YAAY;IACtB,aAAa;IACb,WAAW;GACb;GACA,OAAO,EAAE;EACX,OACE,OAAO,EAAE;CAEb;CACA,OAAO;EACL;EACA,YAAY,YAAY;EACxB,YAAY,SAAS,IAAI,IAAI,QAAQ;EACrC,mBAAmB;EACnB,iBAAiB,WAAW,YAAY;EACxC,gBAAgB;EAChB,gBAAgB;CAClB;AACF;;;;;;;;;;;;;;;;AAgCA,SAAgB,iBACd,kBACA,OAAyD,CAAC,GACrC;CACrB,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,YAAY,KAAK,mBAAmB;CAC1C,MAAM,OAAO,CAAC,GAAG,iBAAiB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,YAAY;EAAE;EAAO;CAAM,EAAE;CACvF,MAAM,UAA+B,CAAC;CAEtC,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAC/B,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACxC,MAAM,IAAI,KAAK;EACf,MAAM,IAAI,KAAK;EACf,MAAM,SAAS,KAAK,IAAI,EAAE,MAAM,QAAQ,EAAE,MAAM,MAAM;EACtD,IAAI,SAAS,YAAY,GAAG;EAO5B,IAAI,gBAAgB;EACpB,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,KAAK;GAC/B,MAAM,KAAK,EAAE,MAAM;GACnB,MAAM,KAAK,EAAE,MAAM;GACnB,MAAM,uBAAuB,GAAG,SAAS,GAAG,QAAQ,GAAG,SAAS,GAAG;GACnE,MAAM,YAAY,KAAK,IAAI,GAAG,SAAS,GAAG,MAAM;GAChD,IAAI,wBAAwB,aAAa,WAAW;IAClD,gBAAgB;IAChB;GACF;EACF;EACA,IAAI,gBAAgB,GAAG;EACvB,IAAI,gBAAgB,WAAW;EAE/B,MAAM,QAAQ,EAAE,MAAM;EACtB,MAAM,QAAQ,EAAE,MAAM;EAEtB,IADe,KAAK,IAAI,MAAM,SAAS,MAAM,MACpC,IAAI,WAAW;EAExB,MAAM,SAAS,MAAM,SAAS,MAAM,SAAS,QAAQ;EACrD,MAAM,WAAW,MAAM,SAAS,MAAM,SAAS,QAAQ;EACvD,MAAM,YAAY,MAAM,SAAS,MAAM,SAAS,EAAE,QAAQ,EAAE;EAC5D,MAAM,cAAc,MAAM,SAAS,MAAM,SAAS,EAAE,QAAQ,EAAE;EAC9D,QAAQ,KAAK;GACX,aAAa;GACb,iBAAiB,gBAAgB;GACjC,cAAc,OAAO;GACrB,cAAc,OAAO;GACrB,gBAAgB,SAAS;GACzB,gBAAgB,SAAS;GACzB,eAAe;GACf,aAAa,OAAO,SAAS,SAAS;EACxC,CAAC;CACH;CAEF,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;AC9GA,eAAsB,cAAiB,MAA0D;CAC/F,MAAM,WAAW,KAAK,YAAY;CAGlC,MAAM,WAAW,MAAM,gBAAgB;EAAE,GAAG;EAAM;CAAS,CAAC;CAG5D,MAAM,gBAAgB,oCACpB,SAAS,MACT,KAAK,oBAAoB,CAAC,CAC5B;CAIA,MAAM,EAAE,MAAM,iBAAiB,MAAM,gBADlB,SAAS,KAAK,QAAQ,QAAQ,aAAa,GAAG,MAAM,KAAA,CACT,GAAG,IAAI,mBAAmB,CAAC;CACzF,MAAM,cAAc,mBAAmB,cAAc;EACnD,GAAG,KAAK;EACR,UAAU,KAAK,aAAa,YAAY;EACxC,WAAW,KAAK,aAAa,aAAa;EAC1C,OAAO,KAAK,aAAa,SAAS;CACpC,CAAC;CAGD,IAAI,oBAAqD;CACzD,IAAI,gBAAuC,CAAC;CAC5C,MAAM,mBAAmB,KAAK,YAAY,oBAAoB;CAC9D,IAAI,EAAE,OAAO,SAAS,gBAAgB,KAAK,oBAAoB,KAAK,oBAAoB,IACtF,MAAM,IAAI,MACR,uFAAuF,kBACzF;CAEF,IAAI,KAAK,QAAQ,YAAY;EAC3B,MAAM,aAAa,KAAK,OAAO;EAC/B,MAAM,SAAS,yBAAyB,SAAS,MAAM,SAAS,YAAY,UAAU;EACtF,gBAAgB,OAAO,KAAK,MAAM,EAAE,QAAQ;EAK5C,IADgB,OAAO,OAAO,MAAM,EAAE,SAAS,YAAY,gBACjD,KAAK,OAAO,MAAM,MAAM,EAAE,OAAO,SAAS,CAAC,GACnD,oBAAoB,iCAAiC;GACnD,aAAa,OAAO,KAAK,EAAE,aAAa,cAAc;IAAE;IAAa;GAAO,EAAE;GAC9E,OAAO,KAAK,YAAY;GACxB,OAAO,KAAK,YAAY;GACxB,MAAM,KAAK,YAAY,QAAQ,KAAK,QAAQ;EAC9C,CAAC;CAEL;CAGA,MAAM,gBAAgB,oBAAoB;EACxC,MAAM,SAAS;EACf,yBAAyB,KAAK;CAChC,CAAC;CAGD,IAAI,qBAA4D;CAChE,IAAI,KAAK,gBAAgB,KAAK,kBAAkB,KAAK,eAAe,SAAS,GAC3E,qBAAqB,MAAM,yBAAyB;EAClD,MAAM,SAAS;EACf,UAAU,KAAK;EACf,gBAAgB,KAAK;CACvB,CAAC;CAIH,MAAM,cAA+C,CAAC;CACtD,IAAI,KAAK,eAAe,KACtB,YAAY,MAAM,MAAM,UAAU,YAAY,OAAO,KAAK,cAAc,KAAK,EAC3E,OAAO,aACT,CAAC;CAEH,IAAI,KAAK,eAAe,MACtB,YAAY,OAAO,MAAM,WAAW,cAAc,KAAK,cAAc,IAAI;CAE3E,IAAI,KAAK,eAAe,KACtB,YAAY,MAAM,MAAM,UAAU,cAAc,KAAK,cAAc,GAAG;CAGxE,MAAM,UAAU,aAAa;EAC3B;EACA;EACA;EACA;EACA;EACA;CACF,CAAC;CAED,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,MAAM;CACR;AACF;;;;;;;;;;;;;;;;;AAoBA,SAAS,yBACP,MACA,YACA,YACiF;CACjF,MAAM,WAAW,MAA4C,GAAG,EAAE,WAAW,IAAI,EAAE;CAMnF,MAAM,oCAAoB,IAAI,IAAyB;CACvD,KAAK,MAAM,KAAK,YAAY;EAC1B,IAAI,MAAM,kBAAkB,IAAI,EAAE,SAAS;EAC3C,IAAI,CAAC,KAAK;GACR,sBAAM,IAAI,IAAY;GACtB,kBAAkB,IAAI,EAAE,WAAW,GAAG;EACxC;EACA,IAAI,IAAI,QAAQ,CAAC,CAAC;CACpB;CAEA,MAAM,kBAAkB,IAAI,IAAY,kBAAkB,IAAI,UAAU,KAAK,CAAC,CAAC;CAC/E,MAAM,kCAAkB,IAAI,IAAoB;CAChD,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,EAAE,gBAAgB,YAAY;EAClC,MAAM,MAAM,QAAQ,CAAC;EACrB,gBAAgB,IAAI,GAAG;EAKvB,MAAM,QAAQ,aAAa,CAAC;EAC5B,IAAI,UAAU,KAAA,GAAW;EACzB,gBAAgB,IAAI,KAAK,KAAK;CAChC;CACA,MAAM,mCAAmB,IAAI,IAAyB;CACtD,MAAM,mCAAmB,IAAI,IAAiC;CAC9D,KAAK,MAAM,CAAC,WAAW,UAAU,mBAAmB;EAClD,IAAI,cAAc,YAAY;EAC9B,iBAAiB,IAAI,WAAW,IAAI,IAAI,KAAK,CAAC;CAChD;CACA,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,EAAE,gBAAgB,YAAY;EAClC,MAAM,MAAM,QAAQ,CAAC;EACrB,IAAI,QAAQ,iBAAiB,IAAI,EAAE,WAAW;EAC9C,IAAI,CAAC,OAAO;GACV,wBAAQ,IAAI,IAAY;GACxB,iBAAiB,IAAI,EAAE,aAAa,KAAK;EAC3C;EACA,MAAM,IAAI,GAAG;EACb,MAAM,QAAQ,aAAa,CAAC;EAC5B,IAAI,UAAU,KAAA,GAAW;EACzB,IAAI,SAAS,iBAAiB,IAAI,EAAE,WAAW;EAC/C,IAAI,CAAC,QAAQ;GACX,yBAAS,IAAI,IAAoB;GACjC,iBAAiB,IAAI,EAAE,aAAa,MAAM;EAC5C;EACA,OAAO,IAAI,KAAK,KAAK;CACvB;CAEA,OAAO,CAAC,GAAG,iBAAiB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,aAAa,oBAAoB;EAC5E,MAAM,SAAS,iBAAiB,IAAI,WAAW,qBAAK,IAAI,IAAoB;EAC5E,MAAM,SAAmB,CAAC;EAC1B,IAAI,oBAAoB;EACxB,IAAI,qBAAqB;EACzB,IAAI,YAAY;EAGhB,KAAK,MAAM,uBAAO,IAAI,IAAI,CAAC,GAAG,gBAAgB,GAAG,eAAe,CAAC,GAAG;GAClE,MAAM,cAAc,eAAe,IAAI,GAAG;GAC1C,MAAM,eAAe,gBAAgB,IAAI,GAAG;GAC5C,IAAI,CAAC,eAAe,CAAC,cAAc;IACjC,aAAa;IACb;GACF;GACA,MAAM,IAAI,OAAO,IAAI,GAAG;GACxB,MAAM,IAAI,gBAAgB,IAAI,GAAG;GACjC,IAAI,MAAM,KAAA,GAAW,qBAAqB;QACrC,IAAI,MAAM,KAAA,GAAW,sBAAsB;QAC3C,OAAO,KAAK,IAAI,CAAC;EACxB;EACA,MAAM,yBAAQ,IAAI,IAAI,CAAC,GAAG,gBAAgB,GAAG,eAAe,CAAC,EAAA,CAAE;EAC/D,OAAO;GACL;GACA;GACA,UAAU;IACR;IACA;IACA,UAAU,OAAO;IACjB;IACA;IACA;IACA,UAAU,UAAU,IAAI,IAAI,OAAO,SAAS;GAC9C;EACF;CACF,CAAC;AACH;AAEA,SAAS,aAAa,MAOX;CACT,MAAM,IAAI,KAAK;CACf,MAAM,QAAQ,CACZ,GAAG,EAAE,WAAW,IAAI,EAAE,KAAK,OAAO,qBAAqB,EAAE,WAAW,OAAO,uBAAuB,EAAE,oBAAoB,MAAM,GAAG,EAAE,EAAE,KACrI,gBAAgB,KAAK,YAAY,MAAM,OAAO,IAAI,KAAK,YAAY,SAAS,IAAI,KAAK,YAAY,iBAAiB,eACpH;CACA,IAAI,KAAK,mBACP,MAAM,KACJ,uBAAuB,KAAK,kBAAkB,eAAe,cAC1D,KAAK,kBAAkB,eAAe,cACnC,IAAI,KAAK,kBAAkB,eAAe,gBAC1C,GACR;CAIF,MAAM,YAAY,KAAK,cAAc,QAAQ,MAAM,EAAE,WAAW,EAAE,KAAK;CACvE,IAAI,UAAU,SAAS,GACrB,MAAM,KACJ,0BAA0B,UACvB,KAAK,MAAM,GAAG,EAAE,YAAY,GAAG,EAAE,SAAS,GAAG,EAAE,OAAO,CAAC,CACvD,KAAK,IAAI,IAAI,KAAK,oBAAoB,KAAK,kCAChD;CAEF,MAAM,KACJ,mBAAmB,KAAK,cAAc,QAAQ,IAAI,KAAK,cAAc,SAAS,OAAO,kBACvF;CACA,IAAI,KAAK,oBAAoB;EAC3B,MAAM,MAAM,KAAK,mBAAmB,OAAO;EAC3C,MAAM,KACJ,eAAe,KAAK,UAAU,OAAO,MAAM,KAAK,YAAY,EAAA,CAAG,QAAQ,CAAC,EAAE,IAAI,KAAK,WAAW,UAAU,EAC1G;CACF;CACA,OAAO,MAAM,KAAK,KAAK;AACzB;;;;;;;;;;;;;;;;;;;;;;;;AC/WA,SAAgB,qBACd,UACA,KACa;CACb,MAAM,WAAW,IAAI,YAAY;CACjC,MAAM,cAAc,IAAI,eAAe,SAAS;CAChD,OAAO,SAAS,MAAM,KAAK,SACzB,wBAAwB,MAAM;EAC5B,OAAO,KAAK;EACZ,cAAc,IAAI;EAClB;EACA,OAAO,IAAI;EACX,YAAY,IAAI;EAChB,YAAY,IAAI;EAChB,WAAW,IAAI;EACf;EACA,gBAAgB,IAAI;CACtB,CAAC,CACH;AACF;;;;;;;;AASA,SAAgB,8BACd,QACA,KACA,OAA2B,CAAC,GACjB;CACX,MAAM,WAAW,IAAI,YAAY;CACjC,MAAM,QAAQ,KAAK,SAAS,OAAO,IAAI,YAAY,GAAG,IAAI,aAAa,GAAG,OAAO;CAEjF,MAAM,YAD2B,OAAO,OAAO,KAAK,uBAE3B,KAAK,aAAa,OAAO,SAAS,IAAI,OAAO,YAAY,KAAA;CAClF,IAAI,sBAAsB;CAC1B,IAAI,kBAAkB;CACtB,IAAI,kBAAkB;CACtB,IAAI,oBAAoB;CACxB,IAAI,qBAAqB;CAEzB,MAAM,MAA8B;EAClC,YAAY,OAAO;EACnB,YAAY,OAAO;EACnB,aAAa,OAAO;EACpB,eAAe,OAAO;EACtB,aAAa,OAAO;EACpB,uBAAuB;CACzB;CACA,KAAK,MAAM,SAAS,OAAO,QAAQ;EACjC,IAAI,wBAAwB,KAAK,GAAG,IAAI,SAAS,MAAM,WAAW,MAAM;OACnE;EACL,IAAI,SAAS,MAAM,MAAM,UAAU,MAAM,WAAW,SAAS,IAAI;EACjE,IAAI,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;GAC1D,IAAI,MAAM,gBAAgB,SAAS;QAC9B;GACL,IAAI,MAAM,WAAW,SAAS;QACzB;EACP;EACA,IAAI,MAAM,aACH;QAAA,MAAM,CAAC,GAAG,MAAM,OAAO,QAAQ,MAAM,WAAW,GACnD,IAAI,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,GAAG,IAAI,SAAS,MAAM,MAAM,GAAG,OAAO;EAAA;CAG1F;CAEA,IAAI,wBAAwB;CAC5B,IAAI,kBAAkB,GAAG,IAAI,oBAAoB;CACjD,IAAI,kBAAkB,GAAG,IAAI,oBAAoB;CACjD,IAAI,oBAAoB,GAAG,IAAI,sBAAsB;CACrD,IAAI,qBAAqB,GAAG,IAAI,uBAAuB;CACvD,IAAI,cAAc,KAAA,GAAW,IAAI,gBAAgB;CAEjD,MAAM,qBAAqB,OAAO,OAAO,MACtC,UAAU,MAAM,WAAW,UAAU,wBAAwB,KAAK,CACrE;CACA,MAAM,UAAgC,EAAE,IAAI;CAC5C,IAAI,cAAc,KAAA,GAChB,IAAI,aAAa,WAAW,QAAQ,eAAe;MAC9C,QAAQ,cAAc;CAG7B,OAAO;EACL;EACA,cAAc,IAAI;EAClB,aAAa,IAAI;EACjB,MAAM;EACN,OAAO,IAAI;EACX,YAAY,IAAI;EAChB,YAAY,IAAI;EAChB,WAAW,IAAI;EACf,QAAQ,OAAO;EACf,SAAS,IAAI,kBAAkB;EAC/B,gBACE,IAAI,mBAAmB,KAAA,IACnB;GAAE,MAAM;GAAc,KAAK;EAAK,IAChC;GAAE,MAAM;GAAa,KAAK,IAAI;EAAe;EACnD,YAAY;GAAE,OAAO;GAAG,QAAQ;EAAE;EAClC,iBAAiB;EACjB;EACA,GAAI,qBACA;GACE,cAAc;GACd,aAAa,SAAS,mBAAmB,MAAM;EACjD,IACA,CAAC;EACL;EACA,YAAY,IAAI;CAClB;AACF;AAEA,SAAS,wBACP,OACmE;CACnE,QAAQ,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,aAAa,MAAM,KAAK;AACzF;AAEA,SAAS,aAAa,OAAiC;CACrD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,KAAK,SAAS,KAAK,SAAS;AACvF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACvHA,MAAa,kBAAkB;;;AAI/B,MAAM,4BAA4B;;;AAIlC,MAAM,2BAA2B;;;;;AAMjC,MAAM,8BAA8B;AAEpC,MAAM,kBAAkB;;;;;;;;;;;;;;;;;;;;;;AAuBxB,MAAa,2BAA6C,WAAW;CACnE,MAAM,MAA8B,OAAO,SAAS,OAAO,CAAC;CAC5D,MAAM,aAAa,aAAa,IAAI,WAAW;CAC/C,MAAM,eAAe,aAAa,IAAI,aAAa;CACnD,MAAM,aAAc,OAAwB;CAC5C,OAAO;EAOL,OACE,aAAa,mBAAmB,QAAQ,SAAS,CAAC,KAClD,aAAa,mBAAmB,QAAQ,QAAQ,CAAC;EACnD,eAAe,OAAO,gBAAgB;EACtC,SAAS,aAAa,OAAO,MAAM;EACnC,eAAe,aAAa,OAAO,YAAY,MAAM;EACrD,YAAY,aAAa,IAAI,eAAe,KAAK,aAAa,IAAI,kBAAkB;EACpF,aAAa;EACb,qBAAqB,kBAAkB,YAAY,cAAc,OAAO,YAAY;EACpF,mBAAmB,OAAO,eAAe,WAAW,WAAW,SAAS;CAC1E;AACF;AAEA,SAAS,kBACP,YACA,cACA,cACe;CACf,IAAI,eAAe,MAAM,OAAO;CAChC,IAAI,eAAe,GAAG,OAAO;CAG7B,QADG,gBAAgB,KAAK,KAAM,iBAAiB,KAAA,KAAa,iBAAiB,YAC7D,gBAAgB;AAClC;AAEA,SAAS,aAAa,OAA+B;CACnD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,IAAI,QAAQ;AACvE;;;;;;;;;AAYA,SAAgB,aAAa,GAA2B,GAAmC;CACzF,MAAM,uBAAO,IAAI,IAAI,CAAC,GAAG,OAAO,KAAK,CAAC,GAAG,GAAG,OAAO,KAAK,CAAC,CAAC,CAAC;CAC3D,IAAI,KAAK,SAAS,GAChB,MAAM,IAAI,gBAAgB,yCAAyC;CAErE,IAAI,OAAO;CACX,IAAI,OAAO;CACX,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,KAAK,EAAE,QAAQ;EACrB,MAAM,KAAK,EAAE,QAAQ;EACrB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,CAAC,OAAO,SAAS,EAAE,KAAK,KAAK,KAAK,KAAK,GACjE,MAAM,IAAI,gBAAgB,4DAA4D,IAAI,EAAE;EAE9F,QAAQ;EACR,QAAQ;CACV;CACA,IAAI,SAAS,KAAK,SAAS,GACzB,MAAM,IAAI,gBAAgB,oEAAoE;CAEhG,IAAI,aAAa;CACjB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,MAAM,EAAE,QAAQ,KAAK;EAC3B,MAAM,MAAM,EAAE,QAAQ,KAAK;EAC3B,MAAM,KAAK,KAAK,MAAM;EACtB,IAAI,KAAK,GAAG,cAAc,KAAM,KAAK,KAAK,KAAK,KAAK,CAAC;EACrD,IAAI,KAAK,GAAG,cAAc,KAAM,KAAK,KAAK,KAAK,KAAK,CAAC;CACvD;CAEA,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,UAAU,CAAC;AAC5C;;;;;;;AAQA,SAAgB,cAAc,QAAkB,cAAc,0BAAoC;CAChG,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,CAAC,OAAO,UAAU,WAAW,KAAK,cAAc,GAClD,MAAM,IAAI,gBACR,2DAA2D,aAC7D;CAEF,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,QAAkB,CAAC;CACzB,KAAK,IAAI,IAAI,GAAG,IAAI,aAAa,KAAK;EACpC,MAAM,MAAO,IAAI,eAAgB,OAAO,SAAS;EACjD,MAAM,KAAK,OAAO,KAAK,MAAM,GAAG;EAChC,MAAM,KAAK,OAAO,KAAK,KAAK,GAAG;EAC/B,MAAM,KAAK,MAAM,MAAM,KAAK,MAAM,GAAG,MAAM,KAAK,GAAG;CACrD;CACA,OAAO,CAAC,GAAG,IAAI,IAAI,KAAK,CAAC;AAC3B;;;AAIA,SAAgB,YAAY,OAAe,OAAyB;CAClE,IAAI,IAAI;CACR,OAAO,IAAI,MAAM,UAAU,SAAS,MAAM,IAAK;CAG/C,OAAO,IAFI,MAAM,IAAI,SAAS,OAAO,MAAM,IAAI,EAAG,EAEpC,GADH,MAAM,MAAM,SAAS,SAAS,OAAO,MAAM,EAAG,EACrC;AACtB;;;;;;;AAuDA,SAAgB,kBACd,WACA,YACA,OAA2B,CAAC,GACZ;CAChB,IAAI,UAAU,WAAW,GACvB,MAAM,IAAI,gBAAgB,gDAAgD;CAE5E,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,gBAAgB,iDAAiD;CAE7E,MAAM,UAAU,KAAK,YAAY;CACjC,MAAM,OAAO,KAAK,kBAAkB;CAEpC,MAAM,UAAU,UAAU,IAAI,OAAO;CACrC,MAAM,WAAW,WAAW,IAAI,OAAO;CAIvC,MAAM,eAAyB,CAAC;CAChC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,OAAO,CAAC,GAAG,SAAS,GAAG,QAAQ,GACxC,KAAK,MAAM,QAAQ,OAAO,KAAK,GAAG,GAChC,IAAI,CAAC,KAAK,IAAI,IAAI,GAAG;EACnB,KAAK,IAAI,IAAI;EACb,aAAa,KAAK,IAAI;CACxB;CAIJ,MAAM,eAAoC,CAAC;CAC3C,MAAM,mBAA6B,CAAC;CAEpC,KAAK,MAAM,WAAW,cAAc;EAClC,MAAM,UAAU,QAAQ,KAAK,MAAM,EAAE,YAAY,IAAI;EACrD,MAAM,WAAW,SAAS,KAAK,MAAM,EAAE,YAAY,IAAI;EACvD,MAAM,OAAO,QAAQ,QAAQ,MAAM,MAAM,IAAI,CAAC,CAAC;EAC/C,MAAM,QAAQ,SAAS,QAAQ,MAAM,MAAM,IAAI,CAAC,CAAC;EACjD,IAAI,OAAO,QAAQ,QAAQ,MAAM;GAC/B,iBAAiB,KAAK,OAAO;GAC7B;EACF;EACA,MAAM,EAAE,KAAK,SAAS,WAAW,SAAS,SAAS,QAAQ;EAC3D,aAAa,KAAK;GAChB;GACA,YAAY,aAAa,KAAK,IAAI;GAClC,WAAW,UAAU,KAAK,QAAQ,QAAQ,MAAM,SAAS,MAAM;GAC/D;GACA;EACF,CAAC;CACH;CAEA,IAAI,aAAa,WAAW,GAC1B,OAAO;EAAE;EAAc,UAAU;EAAY;EAAkB,SAAS;CAAoB;CAE9F,MAAM,WAAW,IAAI,aAAa,QAAQ,KAAK,MAAM,MAAM,EAAE,YAAY,CAAC,IAAI,aAAa;CAC3F,OAAO;EACL;EACA;EACA;EACA,SAAS,YAAY,8BAA8B,mBAAmB;CACxE;AACF;AAIA,SAAS,WACP,SACA,SACA,UAC+D;CAC/D,MAAM,wBAAQ,IAAI,IAAY;CAC9B,KAAK,MAAM,KAAK,CAAC,GAAG,SAAS,GAAG,QAAQ,GACtC,IAAI,MAAM,MAAM,MAAM,IAAI,OAAO,CAAC;CAEpC,IAAI,MAAM,OAAO,GACf,MAAM,IAAI,gBACR,+BAA+B,QAAQ,iFACzC;CAEF,IAAI;CACJ,IAAI,MAAM,IAAI,QAAQ,GAAG;EACvB,MAAM,QAAkB,CAAC;EACzB,KAAK,MAAM,KAAK,CAAC,GAAG,SAAS,GAAG,QAAQ,GACtC,IAAI,MAAM,MAAM,MAAM,KAAK,CAAW;EAExC,MAAM,QAAQ,cAAc,KAAK;EACjC,cAAc,MAAM,YAAY,GAAa,KAAK;CACpD,OACE,cAAc,MAAM;CAEtB,MAAM,SAAS,SAAiD;EAC9D,MAAM,OAA+B,CAAC;EACtC,KAAK,MAAM,KAAK,MAAM;GACpB,MAAM,MAAM,MAAM,OAAO,kBAAkB,WAAW,CAAC;GACvD,KAAK,QAAQ,KAAK,QAAQ,KAAK;EACjC;EACA,OAAO;CACT;CACA,OAAO;EAAE,KAAK,MAAM,OAAO;EAAG,MAAM,MAAM,QAAQ;CAAE;AACtD;AAEA,SAAS,UACP,KACA,UACA,MACA,WACgB;CAEhB,MAAM,SAAS,CADD,mBAAG,IAAI,IAAI,CAAC,GAAG,OAAO,KAAK,GAAG,GAAG,GAAG,OAAO,KAAK,IAAI,CAAC,CAAC,CAClD,CAAC,CAAC,KAAK,WAAW;EAClC;EACA,OAAO,IAAI,UAAU,KAAK;EAC1B,QAAQ,KAAK,UAAU,KAAK;CAC9B,EAAE;CACF,OAAO,MAAM,GAAG,MAAM;EACpB,MAAM,QAAQ,KAAK,IAAI,EAAE,OAAO,EAAE,KAAK,IAAI,KAAK,IAAI,EAAE,OAAO,EAAE,KAAK;EACpE,OAAO,UAAU,IAAI,QAAQ,EAAE,MAAM,cAAc,EAAE,KAAK;CAC5D,CAAC;CACD,OAAO,OAAO,MAAM,GAAG,eAAe;AACxC;;;;;;;;AA8BA,SAAgB,cACd,WACA,YACA,OAAwB,CAAC,GACT;CAChB,IAAI,UAAU,WAAW,GACvB,MAAM,IAAI,gBAAgB,4CAA4C;CAExE,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,gBAAgB,6CAA6C;CAEzE,MAAM,YAAY,KAAK,iBAAiB;CACxC,MAAM,YAAY,KAAK,sBAAsB;CAC7C,MAAM,YAAY,SAAsB,SAAyB;EAC/D,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,SAAS;GAIvB,MAAM,QACJ,aAAa,mBAAmB,GAAG,SAAS,CAAC,KAC7C,aAAa,mBAAmB,GAAG,QAAQ,CAAC;GAC9C,IAAI,UAAU,MACZ,MAAM,IAAI,gBACR,kBAAkB,KAAK,QAAQ,EAAE,MAAM,+CACzC;GAEF,IAAI,SAAS,WAAW;EAC1B;EACA,OAAO,SAAS,QAAQ;CAC1B;CACA,MAAM,cAAc,SAAS,WAAW,WAAW;CACnD,MAAM,eAAe,SAAS,YAAY,YAAY;CACtD,MAAM,MAAM,cAAc;CAC1B,OAAO;EAAE;EAAa;EAAc;EAAK,UAAU,MAAM;CAAU;AACrE;;;;;;;;;;;;AC9WA,SAAgB,gBACd,UACA,OAA2E,CAAC,GAC3D;CACjB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,UAAU,KAAK,iBAAiB;CAKtC,MAAM,YAAY,KAAK,aAAa;CAEpC,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,UAAU;EACxB,WAAW,IAAI,EAAE,MAAM;EACvB,WAAW,IAAI,EAAE,KAAK;CACxB;CACA,MAAM,MAAM,CAAC,GAAG,UAAU,CAAC,CAAC,KAAK;CACjC,MAAM,MAAM,IAAI,IAAI,IAAI,KAAK,IAAI,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC;CAC/C,MAAM,IAAI,IAAI;CACd,IAAI,MAAM,GAAG,OAAO;EAAE,SAAS,CAAC;EAAG,YAAY;EAAG,YAAY;EAAG,WAAW;CAAK;CACjF,IAAI,MAAM,GACR,OAAO;EACL,SAAS,CAAC;GAAE,aAAa,IAAI;GAAK,UAAU;GAAG,aAAa;GAAG,GAAG;GAAG,MAAM;EAAE,CAAC;EAC9E,YAAY;EACZ,YAAY;EACZ,WAAW;CACb;CAKF,MAAM,IAAgB,MAAM,KAAK,EAAE,QAAQ,EAAE,SAAS,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;CAClF,MAAM,IAAgB,MAAM,KAAK,EAAE,QAAQ,EAAE,SAAS,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;CAClF,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,IAAI,IAAI,IAAI,EAAE,MAAM;EAC1B,MAAM,IAAI,IAAI,IAAI,EAAE,KAAK;EACzB,MAAM,IAAI,EAAE,UAAU;EACtB,IAAI,EAAE,MAAM;GACV,EAAE,EAAE,CAAE,MAAO,KAAM;GACnB,EAAE,EAAE,CAAE,MAAO,KAAM;EACrB,OACE,EAAE,EAAE,CAAE,MAAO;EAEf,EAAE,EAAE,CAAE,MAAO;EACb,EAAE,EAAE,CAAE,MAAO;CACf;CAGA,MAAM,YAAY,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;EAC1B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,UAAU,MAAO,EAAE,EAAE,CAAE;EACnD,UAAU,MAAO;CACnB;CACA,MAAM,aAAa,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC;CAC9C,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KACrB,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,WAAW,MAAO,EAAE,EAAE,CAAE;CAItD,IAAI,QAAQ,IAAI,MAAc,CAAC,CAAC,CAAC,KAAK,CAAC;CACvC,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,OAAO,OAAO,SAAS,QAAQ;EAC7B,MAAM,WAAW,IAAI,MAAc,CAAC;EACpC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,IAAI,QAAQ;GACZ,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;IAC1B,IAAI,MAAM,GAAG;IACb,IAAI,EAAE,EAAE,CAAE,OAAQ,GAAG;IACrB,SAAS,EAAE,EAAE,CAAE,MAAO,MAAM,KAAM,MAAM;GAC1C;GACA,SAAS,KAAK,UAAU,IAAI,MAAM,KAAM,UAAU,KAAM;EAC1D;EAEA,IAAI,SAAS;EACb,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,UAAU,KAAK,IAAI,KAAK,IAAI,QAAQ,SAAS,EAAG,CAAC;EAC7E,MAAM,OAAO,KAAK,IAAI,SAAS,CAAC;EAChC,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK,SAAS,KAAK,SAAS,KAAM;EAEzD,QAAQ;EACR,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,IAAI,KAAK,IAAI,SAAS,KAAM,MAAM,EAAG,IAAI,KAAK,IAAI,OAAO,MAAM,EAAG;GACxE,IAAI,IAAI,OAAO,QAAQ;EACzB;EACA,QAAQ;EACR,IAAI,QAAQ,KAAK;CACnB;CAEA,MAAM,SAAS,KAAK,IAAI,GAAG,MAAM,KAAK,MAAM,KAAK,IAAI,KAAK,IAAI,QAAQ,CAAC,CAAC,CAAC,CAAC;CAS1E,OAAO;EACL,SAToC,IAAI,KAAK,IAAI,OAAO;GACxD,aAAa;GACb,UAAU,MAAM;GAChB,aAAa,KAAK,IAAI,KAAK,IAAI,QAAQ,MAAM,EAAG,CAAC,IAAI;GACrD,GAAG,WAAW;GACd,MAAM,UAAU,KAAM;EACxB,EAGiB,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;EACvD,YAAY;EACZ,YAAY;EACZ,WAAW,QAAQ;CACrB;AACF;AAiBA,SAAgB,eACd,SACA,SACA,OAAmB,CAAC,GACyB;CAC7C,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,IAAI,KAAK,WAAW;CAE1B,MAAM,KAAK,QAAQ,IAAI,QAAQ,MAAM,KAAK;CAC1C,MAAM,KAAK,QAAQ,IAAI,QAAQ,KAAK,KAAK;CAEzC,MAAM,YAAY,KAAK,IAAI,QAAQ,KAAK,MAAM;CAC9C,MAAM,SAAS,QAAQ,OAAO,KAAM;CACpC,MAAM,SAAS,QAAQ,OAAO,KAAM;CACpC,MAAM,IAAI,QAAQ,UAAU;CAE5B,MAAM,cAAc,IAAI,KAAK,SAAS;CACtC,MAAM,aAAa,IAAI,KAAK,UAAU,IAAI;CAE1C,QAAQ,IAAI,QAAQ,QAAQ,KAAK,WAAW;CAC5C,QAAQ,IAAI,QAAQ,OAAO,KAAK,UAAU;CAE1C,OAAO;EAAE;EAAa;CAAW;AACnC;AAsBA,SAAgB,0BACd,OACmB;CACnB,MAAM,aAAa,MAAM,cAAc;CACvC,MAAM,wBAAQ,IAAI,IAA2D;CAC7E,KAAK,MAAM,KAAK,MAAM,MAAM;EAC1B,MAAM,MAAM,MAAM,IAAI,EAAE,QAAQ,KAAK,CAAC;EACtC,IAAI,KAAK;GAAE,aAAa,EAAE;GAAa,OAAO,EAAE;EAAM,CAAC;EACvD,MAAM,IAAI,EAAE,UAAU,GAAG;CAC3B;CACA,MAAM,WAA8B,CAAC;CACrC,KAAK,MAAM,OAAO,MAAM,OAAO,GAC7B,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAC9B,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACvC,MAAM,IAAI,IAAI;EACd,MAAM,IAAI,IAAI;EACd,IAAI,EAAE,gBAAgB,EAAE,aAAa;EACrC,MAAM,SAAS,KAAK,IAAI,EAAE,QAAQ,EAAE,KAAK;EACzC,IAAI,UAAU,YACZ,SAAS,KAAK;GAAE,QAAQ,EAAE;GAAa,OAAO,EAAE;GAAa,MAAM;GAAM,QAAQ;EAAE,CAAC;OAC/E;GACL,MAAM,CAAC,QAAQ,SAAS,EAAE,QAAQ,EAAE,QAAQ,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC;GAC1D,SAAS,KAAK;IAAE,QAAQ,OAAO;IAAa,OAAO,MAAM;IAAa,QAAQ;GAAO,CAAC;EACxF;CACF;CAGJ,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;ACzOA,MAAa,0BAA0B;AA4LvC,MAAM,gCAAgC;AAiBtC,SAAS,KAAK,QAAgB,SAAwB;CACpD,MAAM,IAAI,MAAM,sBAAsB,OAAO,IAAI,SAAS;AAC5D;AAEA,SAAS,iBAAiB,QAAgB,WAAgD;CACxF,MAAM,MAAM,UAAU;CACtB,IAAI,QAAQ,MAAM;EAChB,IAAI,UAAU,UACZ,KACE,QACA,2FACF;EAEF,OAAO;GACL,SAAS;GACT,SAAS;GACT,UAAU;GACV,UAAU;GACV,iBAAiB;EACnB;CACF;CACA,MAAM,OAAO;EACX,SAAS,IAAI;EACb,UAAU,IAAI;EACd,UAAU,IAAI;EACd,iBAAiB,IAAI;CACvB;CACA,IAAI,IAAI,YAAY,MAAM;EACxB,IAAI,IAAI,oBAAoB,MAC1B,KACE,QACA,oFACF;EAEF,OAAO;GAAE,SAAS,IAAI,kBAAkB,YAAY;GAAe,GAAG;EAAK;CAC7E;CACA,IAAI,IAAI,aAAa,MAAM,OAAO;EAAE,SAAS;EAAqB,GAAG;CAAK;CAC1E,IAAI,CAAC,IAAI,aAAa,IAAI,YAAY,OAAO;EAAE,SAAS;EAAiB,GAAG;CAAK;CACjF,KAAK,QAAQ,oEAAoE;AACnF;AAEA,SAAS,oBACP,aACA,UACmF;CACnF,IAAI,gBAAgB,QAAQ,gBAAgB,KAAA,GAC1C,OAAO;EAAE,aAAa;EAAM,sBAAsB;EAAO,kBAAkB;CAAE;CAE/E,IAAI,YAAY,UAAU,UACxB,OAAO;EAAE;EAAa,sBAAsB;EAAO,kBAAkB,YAAY;CAAO;CAE1F,OAAO;EACL,aAAa,YAAY,MAAM,GAAG,QAAQ;EAC1C,sBAAsB;EACtB,kBAAkB,YAAY;CAChC;AACF;;;;;AAMA,SAAgB,wBAAwB,MAAuD;CAC7F,MAAM,EAAE,WAAW,OAAO,OAAO,WAAW;CAC5C,MAAM,SAAS,GAAG,KAAK,MAAM,GAAG,UAAU,OAAO,GAAG,UAAU;CAC9D,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,IAAI,UAAU,WAAW,MACvB,KACE,QACA,mBAAmB,UAAU,OAAO,YAAY,UAAU,SAAS,OAAO,uBAC5E;CAEF,IAAI,MAAM,YAAY,UAAU,QAC9B,KAAK,QAAQ,kBAAkB,MAAM,QAAQ,0BAA0B;CAEzE,IAAI,MAAM,eAAe,UAAU,WACjC,KAAK,QAAQ,oBAAoB,MAAM,WAAW,qBAAqB,UAAU,WAAW;CAE9F,IAAI,MAAM,WAAW,UAAU,WAC7B,KAAK,QAAQ,kBAAkB,MAAM,OAAO,uBAAuB,UAAU,WAAW;CAE1F,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;EACrC,MAAM,OAAO,MAAM;EACnB,IAAI,KAAK,YAAY,IAAI,GACvB,KAAK,QAAQ,4CAA4C,EAAE,eAAe,KAAK,SAAS;CAE5F;CACA,MAAM,IAAI,UAAU;CACpB,IAAI,IAAI,KAAK,IAAI,UAAU,WACzB,KAAK,QAAQ,eAAe,EAAE,iBAAiB,UAAU,WAAW;CAEtE,IAAI,CAAC,UAAU,mBAAmB,SAAS,CAAC,GAC1C,KACE,QACA,eAAe,EAAE,iCAAiC,UAAU,mBAAmB,KAAK,IAAI,EAAE,EAC5F;CAEF,MAAM,sBAAsB,CAC1B,GAAG,IAAI,IAAI,MAAM,iBAAiB,SAAS,MAAM,EAAE,kBAAkB,CAAC,CACxE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CACtB,KAAK,MAAM,YAAY,UAAU,oBAC/B,IAAI,CAAC,oBAAoB,SAAS,QAAQ,GACxC,KACE,QACA,kBAAkB,SAAS,oEAC7B;CAGJ,IAAI,WAAW,KAAA,GAAW;EACxB,IAAI,OAAO,MAAM,GAAG,KAAK,QAAQ,sBAAsB,OAAO,EAAE,eAAe,GAAG;EAClF,IAAI,OAAO,mBAAmB,UAAU,gBACtC,KACE,QACA,mCAAmC,OAAO,eAAe,aAAa,UAAU,gBAClF;EAEF,IAAI,OAAO,uBAAuB,UAAU,uBAC1C,KACE,QACA,uCAAuC,OAAO,mBAAmB,aAAa,UAAU,uBAC1F;EAEF,IAAI,OAAO,kBAAkB,WAAW,UAAU,mBAChD,KACE,QACA,0BAA0B,OAAO,kBAAkB,OAAO,gCAAgC,UAAU,mBACtG;EAEF,MAAM,UAAU,OAAO,kBAAkB,QAAQ,MAAM,EAAE,SAAS,qBAAqB,CAAC,CAAC;EACzF,IAAI,YAAY,UAAU,2BACxB,KACE,QACA,0BAA0B,QAAQ,uCAAuC,UAAU,2BACrF;CAEJ;CAEA,MAAM,UAAU,MAAM,IAAI;CAC1B,MAAM,kBAAoC,MAAM,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,UAAU;EACzE,QAAQ,KAAK;EACb,QAAQ,KAAK;EACb,GAAG,oBAAoB,KAAK,aAAa,mBAAmB;CAC9D,EAAE;CAEF,OAAO;EACL,QAAQ;EACR;EACA,QAAQ,UAAU;EAClB,QAAQ,UAAU;EAClB,MAAM;GACJ,OAAO,MAAM,SAAS;GACtB,OAAO,MAAM,SAAS;GACtB,UAAU,MAAM,aAAa;GAC7B,YAAY,MAAM,cAAc;GAChC,QAAQ,MAAM;GACd,WAAW,UAAU;EACvB;EACA,MAAM;GACJ,OAAO;GACP,WAAW,QAAQ;GACnB,oBAAoB,CAAC,GAAG,UAAU,kBAAkB,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;GAC1E;GACA,uBAAuB,UAAU;EACnC;EACA,YAAY;GACV,QAAQ;IAAE,OAAO;IAAG,KAAK;GAAE;GAC3B,OAAO;EACT;EACA,cAAc;GACZ,YAAY,UAAU;GACtB,iBAAiB,UAAU;GAC3B,gBAAgB,QAAQ,kBAAkB;GAC1C,WAAW,UAAU;GACrB,gBAAgB,UAAU;GAC1B,mBAAmB,UAAU;GAC7B,qBAAqB,UAAU;GAC/B,4BAA4B,UAAU;GACtC,2BAA2B,UAAU;GACrC,wBAAwB,QAAQ,qBAAqB;GACrD,UAAU,UAAU;GACpB,qBAAqB,UAAU;GAC/B,aAAa,QAAQ,eAAe;GACpC,QAAQ,UAAU;EACpB;EACA,KAAK,iBAAiB,QAAQ,SAAS;EACvC,YAAY;GACV,OAAO,KAAK;GACZ,kBAAkB,KAAK;GACvB,mBAAmB,KAAK;GACxB,YAAY,KAAK;GACjB,cAAc,KAAK;GACnB,WAAW,KAAK;GAChB,aAAa,KAAK;GAClB,OAAO,UAAU;GACjB,cAAc,UAAU;GACxB,KAAK,UAAU;GACf,WAAW,UAAU;GACrB,eAAe,QAAQ,OAAO,YAAY;GAC1C,WAAW,QAAQ,OAAO,QAAQ;EACpC;CACF;AACF;AAEA,SAAgB,0BAA0B,MAAqD;CAC7F,MAAM,UAAmC;EACvC,MAAM,KAAK;EACX,YAAY;EACZ,iBAAiB;EACjB,KAAK;GAAE,SAAS;GAAG,eAAe;GAAG,qBAAqB;GAAG,iBAAiB;EAAE;EAChF,UAAU,CAAC;CACb;CACA,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,aAAa,YAAY,QAAQ;EACzC,IAAI,IAAI,aAAa,cAAc,IAAI,aAAa,iBAAiB,QAAQ;EAC7E,QAAQ,IAAI,IAAI,IAAI,QAAQ;EAC5B,IAAI,SAAS,QAAQ,SAAS,IAAI;EAClC,IAAI,WAAW,KAAA,GAAW;GACxB,SAAS;IAAE,MAAM;IAAG,YAAY;IAAG,YAAY;GAAE;GACjD,QAAQ,SAAS,IAAI,UAAU;EACjC;EACA,OAAO;EACP,IAAI,IAAI,aAAa,YAAY,OAAO;EACxC,IAAI,IAAI,IAAI,YAAY,WAAW,OAAO;CAC5C;CACA,OAAO;AACT;AAEA,SAAgB,wBAAwB,MAAoC;CAC1E,OAAO,KAAK,KAAK,QAAQ,KAAK,UAAU,GAAG,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,KAAK,SAAS,IAAI,OAAO;AACvF;AAiCA,SAAS,OAAO,QAAwB;CACtC,OAAO,WAAW,QAAQ,CAAC,CAAC,OAAO,MAAM,CAAC,CAAC,OAAO,KAAK;AACzD;AAEA,SAAS,SAAS,MAAc,MAAkD;CAChF,IAAI;CACJ,IAAI;EACF,SAAS,aAAa,IAAI;CAC5B,SAAS,OAAO;EACd,MAAM,IAAI,MACR,kCAAkC,KAAK,MAAM,KAAK,IAAK,MAAgB,SACzE;CACF;CACA,IAAI;EACF,OAAO;GAAE,OAAO,KAAK,MAAM,OAAO,SAAS,MAAM,CAAC;GAAG,QAAQ,OAAO,MAAM;EAAE;CAC9E,SAAS,OAAO;EACd,MAAM,IAAI,MACR,sBAAsB,KAAK,MAAM,KAAK,sBAAuB,MAAgB,SAC/E;CACF;AACF;AAYA,MAAM,0CAA0B,IAAI,IAAI,CAAC,uBAAuB,qBAAqB,CAAC;AAEtF,SAAS,sBAAsB,QAAgB,WAA+C;CAC5F,MAAM,OAAO,KAAK,QAAQ,GAAG,UAAU,OAAO,IAAI,UAAU,UAAU,qBAAqB;CAC3F,MAAM,SAAS,SAAS,MAAM,wBAAwB,UAAU,QAAQ,CAAC,CAAC;CAC1E,MAAM,cAAc,OAAO,qBAAqB,CAAC;CAGjD,KAAK,MAAM,cAAc,aACvB,IAAI,CAAC,wBAAwB,IAAI,WAAW,IAAI,GAC9C,MAAM,IAAI,MACR,sBAAsB,KAAK,QAAQ,WAAW,KAAK,wBAC7C,WAAW,KAAK,qBAAqB,CAAC,GAAG,uBAAuB,CAAC,CAAC,KAAK,IAAI,GACnF;CAGJ,OAAO;EACL,GAAG,OAAO;EACV,gBAAgB,OAAO;EACvB,oBAAoB,OAAO;EAC3B,gBAAgB,OAAO,kBAAkB;EACzC,mBAAmB;EACnB,aAAa,OAAO,MAAM,WAAW;EACrC,QAAQ;GACN,UAAU,OAAO,QAAQ,YAAY;GACrC,MAAM,OAAO,QAAQ,QAAQ;EAC/B;CACF;AACF;;;;;;AAOA,SAAgB,4BACd,QACyB;CACzB,MAAM,SAAS,SAAS,OAAO,iBAAiB,cAAc;CAC9D,MAAM,eAAe,OAAO;CAC5B,IAAI,CAAC,MAAM,QAAQ,aAAa,KAAK,KAAK,aAAa,MAAM,WAAW,GACtE,MAAM,IAAI,MAAM,sCAAsC,OAAO,gBAAgB,cAAc;CAE7F,IAAI,OAAO,aAAa,gBAAgB,YAAY,aAAa,YAAY,WAAW,GACtF,MAAM,IAAI,MACR,sCAAsC,OAAO,gBAAgB,oBAC/D;CAGF,MAAM,6BAAa,IAAI,IAAuE;CAC9F,MAAM,oBAAsE,CAAC;CAE7E,MAAM,iBAAiB,WAAmB;EACxC,MAAM,SAAS,OAAO,QAAQ;EAC9B,IAAI,WAAW,KAAA,GACb,MAAM,IAAI,MACR,sDAAsD,OAAO,kDAC/D;EAEF,IAAI,SAAS,WAAW,IAAI,MAAM;EAClC,IAAI,WAAW,KAAA,GAAW;GACxB,MAAM,SAAS,SAAS,OAAO,YAAY,sBAAsB,OAAO,EAAE;GAC1E,MAAM,UAAU,OAAO;GACvB,IAAI,CAAC,MAAM,QAAQ,OAAO,GACxB,MAAM,IAAI,MACR,yCAAyC,OAAO,OAAO,OAAO,WAAW,kBAC3E;GAEF,MAAM,2BAAW,IAAI,IAA4B;GACjD,KAAK,MAAM,SAAS,SAAS;IAC3B,IAAI,SAAS,IAAI,MAAM,OAAO,GAC5B,MAAM,IAAI,MACR,yCAAyC,OAAO,+BAA+B,MAAM,QAAQ,EAC/F;IAEF,SAAS,IAAI,MAAM,SAAS,KAAK;GACnC;GACA,SAAS;IAAE,QAAQ,OAAO;IAAQ;GAAS;GAC3C,WAAW,IAAI,QAAQ,MAAM;GAC7B,kBAAkB,UAAU;IAC1B,YAAY,OAAO;IACnB,cAAc,OAAO;IACrB,aAAa,OAAO;GACtB;EACF;EACA,OAAO;GAAE;GAAQ,GAAG;EAAO;CAC7B;CAEA,MAAM,OAA6B,CAAC;CACpC,KAAK,MAAM,aAAa,aAAa,OAAO;EAC1C,MAAM,EAAE,QAAQ,QAAQ,cAAc,aAAa,cAAc,UAAU,MAAM;EACjF,MAAM,QAAQ,SAAS,IAAI,UAAU,MAAM;EAC3C,IAAI,UAAU,KAAA,GACZ,MAAM,IAAI,MACR,sBAAsB,OAAO,MAAM,GAAG,UAAU,OAAO,GAAG,UAAU,OAAO,sBAAsB,OAAO,YAC1G;EAEF,MAAM,YAAY,KAAK,OAAO,aAAa,cAAc,UAAU,QAAQ,YAAY;EACvF,MAAM,YAAY,SAAS,WAAW,wBAAwB,UAAU,QAAQ;EAChF,MAAM,QAAQ,UAAU;EACxB,IAAI,CAAC,MAAM,QAAQ,KAAK,GACtB,MAAM,IAAI,MAAM,0CAA0C,UAAU,kBAAkB;EAExF,MAAM,SACJ,OAAO,WAAW,KAAA,IAAY,KAAA,IAAY,sBAAsB,OAAO,QAAQ,SAAS;EAC1F,KAAK,KACH,wBAAwB;GACtB;GACA;GACA;GACA,OAAO,OAAO;GACd,kBAAkB,aAAa;GAC/B,mBAAmB,OAAO;GAC1B,YAAY,OAAO;GACnB;GACA;GACA,aAAa,UAAU;GACvB;GACA,qBAAqB,OAAO;EAC9B,CAAC,CACH;CACF;CACA,KAAK,MAAM,GAAG,MAAM,iBAAiB,EAAE,QAAQ,EAAE,MAAM,CAAC;CAExD,OAAO;EACL;EACA,SAAS,0BAA0B,IAAI;EACvC,YAAY;GACV,OAAO,OAAO;GACd,iBAAiB,OAAO;GACxB,mBAAmB,OAAO;GAC1B,kBAAkB,aAAa;GAC/B,SAAS;EACX;CACF;AACF"}
|