@tangle-network/agent-eval 0.150.2 → 0.161.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +149 -1
- package/README.md +7 -3
- package/dist/{active-curriculum-C4mk67HP.js → active-curriculum-CD5TU2yW.js} +3 -13
- package/dist/active-curriculum-CD5TU2yW.js.map +1 -0
- package/dist/{agent-profile-cell-BkcRDikH.d.ts → agent-profile-cell-CTOZJUuE.d.ts} +4 -2
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +19 -36
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DOCa_QrR.d.ts → backend-integrity-DxuQCu_A.d.ts} +4 -3
- package/dist/backend-integrity-DxuQCu_A.d.ts.map +1 -0
- package/dist/{benchmark-BtAWA8nT.d.ts → benchmark-CGPp-kDC.d.ts} +3 -3
- package/dist/{benchmark-BtAWA8nT.d.ts.map → benchmark-CGPp-kDC.d.ts.map} +1 -1
- package/dist/{benchmark-command-BU1Las59.js → benchmark-command-BVtaq_ve.js} +26 -31
- package/dist/benchmark-command-BVtaq_ve.js.map +1 -0
- package/dist/benchmarks/index.d.ts +6 -19
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +4 -4
- package/dist/benchmarks/index.js.map +1 -1
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +9 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-la-gEYNz.js → campaign-BSmOwskD.js} +77 -795
- package/dist/campaign-BSmOwskD.js.map +1 -0
- package/dist/{canonical-D-XsTQ6_.js → canonical-IL-Bu-14.js} +26 -2
- package/dist/canonical-IL-Bu-14.js.map +1 -0
- package/dist/{capture-fetch-BBVFzhkk.d.ts → capture-fetch-CqwsJkkG.d.ts} +3 -3
- package/dist/{capture-fetch-BBVFzhkk.d.ts.map → capture-fetch-CqwsJkkG.d.ts.map} +1 -1
- package/dist/{chat-client-Bvmxedyv.js → chat-client-DlMlAeYI.js} +5 -55
- package/dist/{chat-client-Bvmxedyv.js.map → chat-client-DlMlAeYI.js.map} +1 -1
- package/dist/chat-json-call-6g5sJobJ.js +53 -0
- package/dist/chat-json-call-6g5sJobJ.js.map +1 -0
- package/dist/cli.js +54 -19
- package/dist/cli.js.map +1 -1
- package/dist/{client-LIuo-KPv.js → client-CX7KqIdB.js} +3 -3
- package/dist/client-CX7KqIdB.js.map +1 -0
- package/dist/{client-kPQYT_56.d.ts → client-L9VVPkim.d.ts} +4 -4
- package/dist/{client-kPQYT_56.d.ts.map → client-L9VVPkim.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -27
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +14 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{counterfactual--bpysZF0.d.ts → counterfactual-BaFUWK3H.d.ts} +4 -4
- package/dist/{counterfactual--bpysZF0.d.ts.map → counterfactual-BaFUWK3H.d.ts.map} +1 -1
- package/dist/{counterfactual-lDfCx0Uz.js → counterfactual-D_VWavVm.js} +2 -2
- package/dist/{counterfactual-lDfCx0Uz.js.map → counterfactual-D_VWavVm.js.map} +1 -1
- package/dist/{dataset-CJjKqQfA.d.ts → dataset-DQqhOCPt.d.ts} +5 -4
- package/dist/{dataset-CJjKqQfA.d.ts.map → dataset-DQqhOCPt.d.ts.map} +1 -1
- package/dist/{default-registry-Cw0Ohdoj.d.ts → default-registry-G9CKMNkc.d.ts} +7 -8
- package/dist/{default-registry-Cw0Ohdoj.d.ts.map → default-registry-G9CKMNkc.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CEQWL9Hy.d.ts → define-agent-eval-Dx1JnPEa.d.ts} +26 -6
- package/dist/define-agent-eval-Dx1JnPEa.d.ts.map +1 -0
- package/dist/{define-agent-eval-C8V8sMqP.js → define-agent-eval-h-s-sI-v.js} +15 -9
- package/dist/define-agent-eval-h-s-sI-v.js.map +1 -0
- package/dist/{dspy-rlm-engine-CBYlPvNy.js → dspy-rlm-engine-DptEII26.js} +95 -15
- package/dist/dspy-rlm-engine-DptEII26.js.map +1 -0
- package/dist/{emitter-CPBAhxum.js → emitter-BpYFQPj4.js} +2 -18
- package/dist/emitter-BpYFQPj4.js.map +1 -0
- package/dist/{emitter-DGQGoLyj.d.ts → emitter-D_jYSGRd.d.ts} +4 -20
- package/dist/{emitter-DGQGoLyj.d.ts.map → emitter-D_jYSGRd.d.ts.map} +1 -1
- package/dist/{engine-BLzhNzoY.d.ts → engine-Cu5qD5Fc.d.ts} +9 -11
- package/dist/{engine-BLzhNzoY.d.ts.map → engine-Cu5qD5Fc.d.ts.map} +1 -1
- package/dist/{eval-campaign-CQuZrLR_.js → eval-campaign-BsXWL2-2.js} +17 -28
- package/dist/eval-campaign-BsXWL2-2.js.map +1 -0
- package/dist/{exact-types-ccQAyut1.d.ts → exact-types-qnexxJ1Z.d.ts} +2 -2
- package/dist/{exact-types-ccQAyut1.d.ts.map → exact-types-qnexxJ1Z.d.ts.map} +1 -1
- package/dist/{exec-y-DCLqK7.js → exec-D9WpA2p-.js} +2 -2
- package/dist/exec-D9WpA2p-.js.map +1 -0
- package/dist/experiment/index.d.ts +45 -11
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +30 -11
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-0MhuPArU.d.ts → experiment-tracker-DCO6Cz4s.d.ts} +2 -2
- package/dist/{experiment-tracker-0MhuPArU.d.ts.map → experiment-tracker-DCO6Cz4s.d.ts.map} +1 -1
- package/dist/{exporters-q9iL-2Jf.js → exporters-Df7TgHFv.js} +3 -3
- package/dist/exporters-Df7TgHFv.js.map +1 -0
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts → external-optimizer-contracts-szBJ_1vh.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts.map → external-optimizer-contracts-szBJ_1vh.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-x9oEXKsU.js → external-optimizer-process-WosTBChy.js} +4 -4
- package/dist/{external-optimizer-process-x9oEXKsU.js.map → external-optimizer-process-WosTBChy.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-CKNb42oM.js → external-optimizer-subprocess-BIWbHpgD.js} +4 -4
- package/dist/{external-optimizer-subprocess-CKNb42oM.js.map → external-optimizer-subprocess-BIWbHpgD.js.map} +1 -1
- package/dist/{failure-cluster-BLURuWG4.d.ts → failure-cluster-CXL8NbEw.d.ts} +3 -3
- package/dist/{failure-cluster-BLURuWG4.d.ts.map → failure-cluster-CXL8NbEw.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts → feedback-trajectory-B3ZHaHV_.d.ts} +7 -7
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-XggBupCr.js → hf-dataset-D8_RNIis.js} +4 -4
- package/dist/{hf-dataset-XggBupCr.js.map → hf-dataset-D8_RNIis.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -14
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +2 -2
- package/dist/{index-Aj3WO3_a.d.ts → index-CGtH1piv.d.ts} +48 -26
- package/dist/index-CGtH1piv.d.ts.map +1 -0
- package/dist/{index-BNPtkBPf.d.ts → index-D-IiQIBB.d.ts} +5 -10
- package/dist/index-D-IiQIBB.d.ts.map +1 -0
- package/dist/{index-IQccV3Ou.d.ts → index-D-V8gCs_.d.ts} +13 -90
- package/dist/index-D-V8gCs_.d.ts.map +1 -0
- package/dist/{index-B8Ui1mr1.d.ts → index-lfaSeKSD.d.ts} +18 -2
- package/dist/index-lfaSeKSD.d.ts.map +1 -0
- package/dist/index-vrJugRal.d.ts +1 -0
- package/dist/index.d.ts +68 -56
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +59 -110
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BeT8KCgI.d.ts → insight-report-DRe8LB6d.d.ts} +4 -4
- package/dist/{insight-report-BeT8KCgI.d.ts.map → insight-report-DRe8LB6d.d.ts.map} +1 -1
- package/dist/{integrity-DL91tucI.js → integrity-CyWSSoQS.js} +2 -2
- package/dist/{integrity-DL91tucI.js.map → integrity-CyWSSoQS.js.map} +1 -1
- package/dist/{integrity-B0dZ96EO.d.ts → integrity-DUNX9Fao.d.ts} +3 -3
- package/dist/{integrity-B0dZ96EO.d.ts.map → integrity-DUNX9Fao.d.ts.map} +1 -1
- package/dist/{kind-factory-DmAa0h3K.js → kind-factory-DY8FdoXf.js} +3 -71
- package/dist/kind-factory-DY8FdoXf.js.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +3 -3
- package/dist/{ledger-core-DTae9rv_.js → ledger-core-BOzlRygb.js} +2 -2
- package/dist/{ledger-core-DTae9rv_.js.map → ledger-core-BOzlRygb.js.map} +1 -1
- package/dist/{llm-client-Bg32RW0j.js → llm-client-hgDieDNN.js} +53 -98
- package/dist/llm-client-hgDieDNN.js.map +1 -0
- package/dist/{llm-judge-BqqMS8t7.js → llm-judge-BhasIPFT.js} +1135 -77
- package/dist/llm-judge-BhasIPFT.js.map +1 -0
- package/dist/{matrix-DrVnRp4G.d.ts → matrix-eXKRMHnL.d.ts} +74 -72
- package/dist/matrix-eXKRMHnL.d.ts.map +1 -0
- package/dist/meta-eval/index.d.ts +8 -6
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +8 -6
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-BV6tLVWl.js → mint-DfODW1KW.js} +3 -3
- package/dist/{mint-BV6tLVWl.js.map → mint-DfODW1KW.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +2 -8
- package/dist/multishot/golden/index.d.ts.map +1 -1
- package/dist/multishot/golden/index.js +56 -86
- package/dist/multishot/golden/index.js.map +1 -1
- package/dist/multishot/index.d.ts +11 -46
- package/dist/multishot/index.d.ts.map +1 -1
- package/dist/multishot/index.js +30 -83
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-DJWAXLms.js → opencode-sqlite-eK6HW6dr.js} +2 -6
- package/dist/{opencode-sqlite-DJWAXLms.js.map → opencode-sqlite-eK6HW6dr.js.map} +1 -1
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pre-registration-zFSLEiFU.d.ts → pre-registration-CzFCcwYk.d.ts} +55 -40
- package/dist/pre-registration-CzFCcwYk.d.ts.map +1 -0
- package/dist/pre-registration-KN9jkh58.js +110 -0
- package/dist/pre-registration-KN9jkh58.js.map +1 -0
- package/dist/{produced-state-Bnq4FaDO.js → produced-state-DZ89riy5.js} +8 -8
- package/dist/produced-state-DZ89riy5.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +31 -5
- package/dist/profile-cell.js.map +1 -1
- package/dist/{promotion-policy-DLOUkYhI.d.ts → promotion-policy-DtnOIZvk.d.ts} +2 -2
- package/dist/{promotion-policy-DLOUkYhI.d.ts.map → promotion-policy-DtnOIZvk.d.ts.map} +1 -1
- package/dist/{query-Di7eEQ79.js → query-CHmMP42p.js} +20 -11
- package/dist/query-CHmMP42p.js.map +1 -0
- package/dist/{query-CJ_DX8vl.d.ts → query-DxPYqpmT.d.ts} +10 -4
- package/dist/query-DxPYqpmT.d.ts.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/{registry-BQwrSYpC.d.ts → registry-8You7OK1.d.ts} +5 -7
- package/dist/{registry-BQwrSYpC.d.ts.map → registry-8You7OK1.d.ts.map} +1 -1
- package/dist/{release-confidence-BknrpBnO.js → release-confidence-DKfD2RYU.js} +28 -14
- package/dist/release-confidence-DKfD2RYU.js.map +1 -0
- package/dist/{release-confidence-4XrqlpFD.d.ts → release-confidence-Dqt0NFep.d.ts} +7 -6
- package/dist/release-confidence-Dqt0NFep.d.ts.map +1 -0
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +3 -3
- package/dist/{researcher-DJnoUE8c.d.ts → researcher-Cz565b7D.d.ts} +34 -21
- package/dist/researcher-Cz565b7D.d.ts.map +1 -0
- package/dist/{reward-hacking-62tojkQd.d.ts → reward-hacking-MBf7qpSB.d.ts} +2 -2
- package/dist/{reward-hacking-62tojkQd.d.ts.map → reward-hacking-MBf7qpSB.d.ts.map} +1 -1
- package/dist/{reward-hacking-C0x0xihA.js → reward-hacking-t4lB1yt8.js} +2 -2
- package/dist/{reward-hacking-C0x0xihA.js.map → reward-hacking-t4lB1yt8.js.map} +1 -1
- package/dist/rl.d.ts +11 -42
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +41 -24
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +3 -3
- package/dist/rollout/index.js +7 -7
- package/dist/{rollout-ytVQ7WT8.js → rollout-Dm2tSdiQ.js} +6 -6
- package/dist/{rollout-ytVQ7WT8.js.map → rollout-Dm2tSdiQ.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CzxLoZge.js → rubric-predictive-validity-CK8SCOg-.js} +5 -15
- package/dist/rubric-predictive-validity-CK8SCOg-.js.map +1 -0
- package/dist/{rubric-predictive-validity-C7LnNvF2.d.ts → rubric-predictive-validity-CxycqzX5.d.ts} +4 -3
- package/dist/rubric-predictive-validity-CxycqzX5.d.ts.map +1 -0
- package/dist/{run-record-D2lDdSAz.js → run-record-BC0ebuRP.js} +2 -2
- package/dist/{run-record-D2lDdSAz.js.map → run-record-BC0ebuRP.js.map} +1 -1
- package/dist/{run-record-DVV82Gwh.d.ts → run-record-VVy4T9OW.d.ts} +3 -3
- package/dist/{run-record-DVV82Gwh.d.ts.map → run-record-VVy4T9OW.d.ts.map} +1 -1
- package/dist/{schema-BtVldJ3T.d.ts → schema-Bjgdsn73.d.ts} +2 -4
- package/dist/{schema-BtVldJ3T.d.ts.map → schema-Bjgdsn73.d.ts.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-BzWDXhOR.d.ts} +2 -5
- package/dist/schema-BzWDXhOR.d.ts.map +1 -0
- package/dist/{schema-C6DW4ZHR.js → schema-C1aaAxTf.js} +2 -2
- package/dist/schema-C1aaAxTf.js.map +1 -0
- package/dist/{schema-CRhEY1SO.js → schema-k6ZBftVv.js} +2 -8
- package/dist/{schema-CRhEY1SO.js.map → schema-k6ZBftVv.js.map} +1 -1
- package/dist/{semantic-concept-judge-laMCnTLn.js → semantic-concept-judge-BSkKKHeq.js} +14 -38
- package/dist/semantic-concept-judge-BSkKKHeq.js.map +1 -0
- package/dist/{sequential-eprocess-CbUt2htw.js → sequential-eprocess-D1jKoihe.js} +49 -2
- package/dist/sequential-eprocess-D1jKoihe.js.map +1 -0
- package/dist/{sequential-C458DXNf.js → sequential-rYW-Ophm.js} +41 -16
- package/dist/sequential-rYW-Ophm.js.map +1 -0
- package/dist/{series-convergence-BxKEgBwA.d.ts → series-convergence-D9WgpXGi.d.ts} +2 -2
- package/dist/{series-convergence-BxKEgBwA.d.ts.map → series-convergence-D9WgpXGi.d.ts.map} +1 -1
- package/dist/{server-dIWwF3j_.js → server-BtFd4uzB.js} +19 -42
- package/dist/server-BtFd4uzB.js.map +1 -0
- package/dist/{skillopt-optimization-method-CPBlTcj5.js → skillopt-optimization-method-DbaekMcn.js} +794 -8
- package/dist/skillopt-optimization-method-DbaekMcn.js.map +1 -0
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts → skillopt-optimization-method-x7TTF23P.d.ts} +20 -7
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts.map → skillopt-optimization-method-x7TTF23P.d.ts.map} +1 -1
- package/dist/{statistical-heldout-_woZ9q9j.d.ts → statistical-heldout-Cy3EhjlC.d.ts} +21 -8
- package/dist/statistical-heldout-Cy3EhjlC.d.ts.map +1 -0
- package/dist/{steps-AmkT-GIM.d.ts → steps-CiNVJry_.d.ts} +2 -17
- package/dist/steps-CiNVJry_.d.ts.map +1 -0
- package/dist/{store-CT9YIIve.d.ts → store-B06JdC56.d.ts} +2 -2
- package/dist/{store-CT9YIIve.d.ts.map → store-B06JdC56.d.ts.map} +1 -1
- package/dist/{store-otlp-CDYWW_8N.js → store-otlp-C_Rq5I4D.js} +2 -2
- package/dist/{store-otlp-CDYWW_8N.js.map → store-otlp-C_Rq5I4D.js.map} +1 -1
- package/dist/{store-tool-spans-Br2_IUhm.d.ts → store-tool-spans-DPUG7UUY.d.ts} +6 -6
- package/dist/{store-tool-spans-Br2_IUhm.d.ts.map → store-tool-spans-DPUG7UUY.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CykkbOlv.js → store-tool-spans-Dlh9vkFK.js} +3 -3
- package/dist/{store-tool-spans-CykkbOlv.js.map → store-tool-spans-Dlh9vkFK.js.map} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DW2bEpdB.js → summary-report-BI5hUtvK.js} +6 -16
- package/dist/summary-report-BI5hUtvK.js.map +1 -0
- package/dist/{summary-report-B__Y5ub3.d.ts → summary-report-CC07PhEL.d.ts} +6 -5
- package/dist/summary-report-CC07PhEL.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +3 -16
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes--ZTP3tYO.js → task-failure-attributes-DTl-7-Kw.js} +3 -3
- package/dist/{task-failure-attributes--ZTP3tYO.js.map → task-failure-attributes-DTl-7-Kw.js.map} +1 -1
- package/dist/{tool-groups-B4tqh8jB.d.ts → tool-groups-Ci8i9ErB.d.ts} +3 -3
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +1 -0
- package/dist/{tool-waste-C-VHSRwF.js → tool-waste-BqzmVdJk.js} +2 -2
- package/dist/{tool-waste-C-VHSRwF.js.map → tool-waste-BqzmVdJk.js.map} +1 -1
- package/dist/{tool-waste-DjRDEsuI.d.ts → tool-waste-Dro0gJi3.d.ts} +4 -4
- package/dist/{tool-waste-DjRDEsuI.d.ts.map → tool-waste-Dro0gJi3.d.ts.map} +1 -1
- package/dist/trace-repair/index.d.ts +4 -77
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -15
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +13 -23
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +9 -20
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-YC15QDYQ.d.ts → trajectory-Bi157Gun.d.ts} +3 -3
- package/dist/{trajectory-YC15QDYQ.d.ts.map → trajectory-Bi157Gun.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +5 -5
- package/dist/trajectory-replay/index.js +5 -5
- package/dist/{provenance-oA4-zUqm.d.ts → transient-failure-DKF5Mofa.d.ts} +468 -13
- package/dist/transient-failure-DKF5Mofa.d.ts.map +1 -0
- package/dist/types-B3jzCp0p.js.map +1 -1
- package/dist/{types-CLAwnY-L.d.ts → types-BPb2Kf_C2.d.ts} +3 -3
- package/dist/types-BPb2Kf_C2.d.ts.map +1 -0
- package/dist/types-Bfk0uxRj.d.ts +443 -0
- package/dist/types-Bfk0uxRj.d.ts.map +1 -0
- package/dist/{types-DdFNuyxQ.d.ts → types-D4s7Z6nq.d.ts} +30 -6
- package/dist/types-D4s7Z6nq.d.ts.map +1 -0
- package/dist/{types-B2NsbrNy.d.ts → types-D9ssmxKL.d.ts} +3 -3
- package/dist/{types-B2NsbrNy.d.ts.map → types-D9ssmxKL.d.ts.map} +1 -1
- package/dist/{types-I5WwVzQ7.d.ts → types-DeIUdzNd.d.ts} +2 -2
- package/dist/{types-I5WwVzQ7.d.ts.map → types-DeIUdzNd.d.ts.map} +1 -1
- package/dist/{verdict-BndeTAh_.js → verdict-B0xltqu6.js} +2 -2
- package/dist/{verdict-BndeTAh_.js.map → verdict-B0xltqu6.js.map} +1 -1
- package/dist/verdict-cache-CdVVTVmn.js +88 -0
- package/dist/verdict-cache-CdVVTVmn.js.map +1 -0
- package/dist/wire/index.d.ts +21 -111
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/building-doctrine.md +3 -3
- package/docs/campaign-proposers.md +41 -10
- package/docs/design/statistics-decisions.md +89 -1
- package/docs/eval-surface-map.md +14 -0
- package/docs/experiment.md +19 -2
- package/docs/feedback-trajectories.md +1 -1
- package/docs/multishot-golden-records.md +4 -4
- package/docs/public-api.md +1616 -0
- package/docs/research-report-methodology.md +1 -1
- package/docs/search-history-receipts.md +39 -1
- package/docs/trace-analysis.md +1 -1
- package/docs/trace-repair-admission.md +1 -1
- package/docs/trace-repair-continuation.md +1 -1
- package/docs/verdicts.md +24 -0
- package/package.json +6 -2
- package/dist/active-curriculum-C4mk67HP.js.map +0 -1
- package/dist/agent-profile-cell-BkcRDikH.d.ts.map +0 -1
- package/dist/backend-integrity-DOCa_QrR.d.ts.map +0 -1
- package/dist/benchmark-command-BU1Las59.js.map +0 -1
- package/dist/campaign-la-gEYNz.js.map +0 -1
- package/dist/canonical-D-XsTQ6_.js.map +0 -1
- package/dist/client-LIuo-KPv.js.map +0 -1
- package/dist/define-agent-eval-C8V8sMqP.js.map +0 -1
- package/dist/define-agent-eval-CEQWL9Hy.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-CBYlPvNy.js.map +0 -1
- package/dist/emitter-CPBAhxum.js.map +0 -1
- package/dist/eval-campaign-CQuZrLR_.js.map +0 -1
- package/dist/exec-y-DCLqK7.js.map +0 -1
- package/dist/exporters-q9iL-2Jf.js.map +0 -1
- package/dist/index-Aj3WO3_a.d.ts.map +0 -1
- package/dist/index-B8Ui1mr1.d.ts.map +0 -1
- package/dist/index-BNPtkBPf.d.ts.map +0 -1
- package/dist/index-C1ravkGA.d.ts +0 -1
- package/dist/index-IQccV3Ou.d.ts.map +0 -1
- package/dist/kind-factory-DmAa0h3K.js.map +0 -1
- package/dist/llm-client-Bg32RW0j.js.map +0 -1
- package/dist/llm-judge-BqqMS8t7.js.map +0 -1
- package/dist/matrix-DrVnRp4G.d.ts.map +0 -1
- package/dist/pre-registration-DakwTRXk.js +0 -96
- package/dist/pre-registration-DakwTRXk.js.map +0 -1
- package/dist/pre-registration-zFSLEiFU.d.ts.map +0 -1
- package/dist/produced-state-Bnq4FaDO.js.map +0 -1
- package/dist/provenance-oA4-zUqm.d.ts.map +0 -1
- package/dist/query-CJ_DX8vl.d.ts.map +0 -1
- package/dist/query-Di7eEQ79.js.map +0 -1
- package/dist/release-confidence-4XrqlpFD.d.ts.map +0 -1
- package/dist/release-confidence-BknrpBnO.js.map +0 -1
- package/dist/researcher-DJnoUE8c.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-C7LnNvF2.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-CzxLoZge.js.map +0 -1
- package/dist/schema-C6DW4ZHR.js.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/semantic-concept-judge-laMCnTLn.js.map +0 -1
- package/dist/sequential-C458DXNf.js.map +0 -1
- package/dist/sequential-eprocess-CbUt2htw.js.map +0 -1
- package/dist/server-dIWwF3j_.js.map +0 -1
- package/dist/skillopt-optimization-method-CPBlTcj5.js.map +0 -1
- package/dist/statistical-heldout-_woZ9q9j.d.ts.map +0 -1
- package/dist/steps-AmkT-GIM.d.ts.map +0 -1
- package/dist/summary-report-B__Y5ub3.d.ts.map +0 -1
- package/dist/summary-report-DW2bEpdB.js.map +0 -1
- package/dist/tool-groups-B4tqh8jB.d.ts.map +0 -1
- package/dist/types-CLAwnY-L.d.ts.map +0 -1
- package/dist/types-DdFNuyxQ.d.ts.map +0 -1
- package/dist/types-jUBXJ7Iz.d.ts +0 -884
- package/dist/types-jUBXJ7Iz.d.ts.map +0 -1
- package/dist/verdict-cache-mZf5FEiY.js +0 -107
- package/dist/verdict-cache-mZf5FEiY.js.map +0 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { C as TraceEvent, _ as Span, f as Run, o as FailureClass } from "./schema-
|
|
2
|
-
import { s as TraceStore } from "./store-
|
|
1
|
+
import { C as TraceEvent, _ as Span, f as Run, o as FailureClass } from "./schema-Bjgdsn73.js";
|
|
2
|
+
import { s as TraceStore } from "./store-B06JdC56.js";
|
|
3
3
|
//#region src/failure-taxonomy.d.ts
|
|
4
4
|
interface FailureContext {
|
|
5
5
|
run: Run;
|
|
@@ -55,4 +55,4 @@ declare function failureClusterView(store: TraceStore, options?: {
|
|
|
55
55
|
}): Promise<FailureClusterReport>;
|
|
56
56
|
//#endregion
|
|
57
57
|
export { FailureContext as a, FailureClassification as i, FailureClusterReport as n, FailureRule as o, failureClusterView as r, classifyFailure as s, FailureCluster as t };
|
|
58
|
-
//# sourceMappingURL=failure-cluster-
|
|
58
|
+
//# sourceMappingURL=failure-cluster-CXL8NbEw.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"failure-cluster-
|
|
1
|
+
{"version":3,"file":"failure-cluster-CXL8NbEw.d.ts","names":[],"sources":["../src/failure-taxonomy.ts","../src/pipelines/failure-cluster.ts"],"mappings":";;;UAgBiB;EACf,KAAK;EACL,OAAO;EACP,QAAQ;;UAGO;EACf,cAAc;EACd;EACA;EACA;;;UAIe;EACf;EACA,QAAQ,KAAK;IACX,cAAc;IACd;IACA;IACA;;;;iBA6ZY,gBACd,KAAK,gBACL,QAAO,gBACN;;;UCvbc;EACf,cAAc;;EAEd;;EAEA;;;;;;;EAOA;EACA;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;iBAGoB,mBACpB,OAAO,YACP;EAAW,QAAQ;EAAe;IACjC,QAAQ"}
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { o as FailureClass } from "./schema-
|
|
2
|
-
import { s as TraceStore } from "./store-
|
|
3
|
-
import { i as TraceEmitter } from "./emitter-
|
|
4
|
-
import { i as AnalystFinding, l as AnalystRunResult, p as EvidenceRef } from "./types-
|
|
5
|
-
import "./exact-types-
|
|
6
|
-
import { i as DatasetSplit, r as DatasetScenario } from "./dataset-
|
|
1
|
+
import { o as FailureClass } from "./schema-Bjgdsn73.js";
|
|
2
|
+
import { s as TraceStore } from "./store-B06JdC56.js";
|
|
3
|
+
import { i as TraceEmitter } from "./emitter-D_jYSGRd.js";
|
|
4
|
+
import { i as AnalystFinding, l as AnalystRunResult, p as EvidenceRef } from "./types-D9ssmxKL.js";
|
|
5
|
+
import "./exact-types-qnexxJ1Z.js";
|
|
6
|
+
import { i as DatasetSplit, r as DatasetScenario } from "./dataset-DQqhOCPt.js";
|
|
7
7
|
//#region src/control-runtime.d.ts
|
|
8
8
|
type ControlSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
9
9
|
type ControlActionFailureMode = 'continue' | 'stop';
|
|
@@ -394,4 +394,4 @@ declare function controlRunToFeedbackTrajectory<TState, TAction, TActionResult>(
|
|
|
394
394
|
}): FeedbackTrajectory;
|
|
395
395
|
//#endregion
|
|
396
396
|
export { AnalystFindingDigest as A, ControlRunResult as B, createFeedbackTrajectory as C, renderPreferenceMemoryMarkdown as D, feedbackTrajectoryToOptimizerRow as E, ControlActionOutcome as F, StopDecision as G, ControlRuntimeError as H, ControlBudget as I, subjectiveEval as J, objectiveEval as K, ControlContext as L, AnalystRunDigest as M, analystFindingDigest as N, summarizePreferenceMemory as O, analystRunDigest as P, ControlDecision as R, controlRunToFeedbackTrajectory as S, feedbackTrajectoriesToOptimizerRows as T, ControlSeverity as U, ControlRuntimeConfig as V, ControlStep as W, PreferenceMemoryEntry as _, FeedbackLabel as a, analystRunToReviewRequests as b, FeedbackOptimizerRow as c, FeedbackTask as d, FeedbackTrajectory as f, InMemoryFeedbackTrajectoryStore as g, FileSystemFeedbackTrajectoryStore as h, FeedbackAttempt as i, AnalystReviewDecision as j, withAssignedFeedbackSplit as k, FeedbackOutcome as l, FeedbackTrajectoryStore as m, AnalystReviewRequest as n, FeedbackLabelKind as o, FeedbackTrajectoryFilter as p, runAgentControlLoop as q, FeedbackArtifactType as r, FeedbackLabelSource as s, AnalystFeedbackTrajectoryOptions as t, FeedbackSplitPolicy as u, ProposedSideEffect as v, feedbackTrajectoriesToDatasetScenarios as w, assignFeedbackSplit as x, analystRunToFeedbackTrajectory as y, ControlEvalResult as z };
|
|
397
|
-
//# sourceMappingURL=feedback-trajectory-
|
|
397
|
+
//# sourceMappingURL=feedback-trajectory-B3ZHaHV_.d.ts.map
|
package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map}
RENAMED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"feedback-trajectory-
|
|
1
|
+
{"version":3,"file":"feedback-trajectory-B3ZHaHV_.d.ts","names":[],"sources":["../src/control-runtime.ts","../src/feedback-trajectory-review.ts","../src/feedback-trajectory.ts"],"mappings":";;;;;;;KAkBY;KACA;UAEK;;EAEf;;EAEA;;EAEA;;EAEA,WAAW;;EAEX;;EAEA;;EAEA;;EAEA,WAAW;;UAGI;EACf;EACA;EACA;;UAGe,oBAAoB,QAAQ;;;;;EAK3C;;;;;EAKA;;EAEA;;EAEA,oBAAoB,OAAO;;EAE3B,qBAAqB,QAAQ;;UAGd,eACf,QACA,SACA,eACA,cAAc,oBAAoB;EAElC;EACA,OAAO;EACP,OAAO;EACP,SAAS,YAAY,QAAQ,SAAS,eAAe;EACrD,QAAQ;EACR;EACA;EACA;EACA;EACA,aAAa;EACb,UAAU;;KAGA,gBAAgB;EAEtB;EACA,QAAQ;EACR;;EAGA;EACA;EACA;EACA;;EAEA,eAAe;;UAGJ;EACf;EACA;EACA;EACA;EACA,eAAe;;UAGA,qBAAqB;EACpC;EACA,SAAS;EACT;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe,YACf,QACA,SACA,eACA,cAAc,oBAAoB;EAElC;EACA,UAAU,gBAAgB;EAC1B,aAAa;EACb,YAAY;EACZ,aAAa;EACb,YAAY;EACZ,gBAAgB,qBAAqB;EACrC;EACA;;UAGe,iBACf,QACA,SACA,eACA,cAAc,oBAAoB;EAElC;EACA;EACA;EACA;EACA;EACA,OAAO,YAAY,QAAQ,SAAS,eAAe;EACnD,YAAY;EACZ,YAAY;EACZ;EACA;;EAEA;EACA,eAAe;EACf,eAAe;EACf;;UAGe,qBACf,QACA,SACA,eACA,cAAc,oBAAoB;EAElC;EACA,SAAS,QAAQ;EACjB,SAAS;;EAET,gBAAgB;;;;;EAKhB,oBAAoB;IAClB,QAAQ;IACR,QAAQ;IACR,OAAO;IACP,OAAO;IACP,SAAS,YAAY,QAAQ,SAAS,eAAe;;;EAIvD,UAAU;IACR,SAAS,YAAY,QAAQ,SAAS,eAAe;IACrD,aAAa;QACT,QAAQ,UAAU;;EAGxB,WAAW;IACT;IACA,OAAO;IACP,SAAS,YAAY,QAAQ,SAAS,eAAe;IACrD,aAAa;QACT,QAAQ,WAAW;;EAGzB,SACE,KAAK,eAAe,QAAQ,SAAS,eAAe,WACjD,QAAQ,gBAAgB,YAAY,gBAAgB;;EAGzD,MACE,QAAQ,SACR,KAAK,eAAe,QAAQ,SAAS,eAAe,WACjD,QAAQ,iBAAiB;;EAG9B,cACE,KAAK,eAAe,QAAQ,SAAS,eAAe,WACjD,QAAQ,gBAAgB;;EAG7B,UAAU,MAAM,YAAY,QAAQ,SAAS,eAAe,WAAW;;EAGvE,eAAe,oBAAoB,QAAQ;;EAG3C,QAAQ;EACR;EACA;EACA;;iBAQoB,oBACpB,QACA,SACA,eACA,cAAc,oBAAoB,mBAElC,QAAQ,qBAAqB,QAAQ,SAAS,eAAe,SAC5D,QAAQ,iBAAiB,QAAQ,SAAS,eAAe;iBAinB5C,cAAc,OAAO,KAAK,kCAAkC;iBAI5D,eAAe,OAAO,KAAK,kCAAkC;;;KCx1BjE,sBAAsB,QAAQ;KAC9B;KACA;UAEK;EACf;EACA;EACA,WAAW;;UAGH;EACR,WAAW;EACX,QAAQ;EACR;EACA;EACA;EACA;;KAGU,yBACP;EACC;EACA,eAAe;EACf;EACA;MAED;EACC;EACA,cAAc;EACd;EACA;;;iBA2BU,qBAAqB,SAAS,iBAAiB;;iBAK/C,iBAAiB,KAAK,mBAAmB;;;KChD7C;KAWA;KAEA;KAYA;UAEK;EACf;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA,WAAW;;UAGI;EACf;EACA,QAAQ;EACR,MAAM;EACN;EACA;EACA,WAAW;EACX;EACA,WAAW;;UAGI;EACf;EACA;EACA,cAAc;EACd;EACA;EACA,iBAAiB;EACjB,QAAQ;EACR,WAAW;EACX;EACA,WAAW;;UAGI;EACf;EACA;EACA,UAAU;EACV;EACA;EACA;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA,MAAM;EACN,UAAU;EACV,QAAQ;EACR,UAAU;EACV,QAAQ;EACR,OAAO;EACP;EACA;EACA,WAAW;;UAGI;EACf,KAAK,YAAY,qBAAqB;EACtC,IAAI,aAAa,QAAQ;EACzB,KAAK,SAAS,2BAA2B,QAAQ;EACjD,cAAc,YAAY,SAAS,kBAAkB,QAAQ;EAC7D,YAAY,YAAY,OAAO,eAAe,qBAAqB,QAAQ;;UAG5D;EACf;EACA;EACA,QAAQ;EACR;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,YAAY;EACZ;EACA,WAAW;;UAoBI;EACf,MAAM;EACN,kBAAkB;EAClB,0BAA0B;EAC1B,2BAA2B;EAC3B,UAAU;EACV;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP;EACA;IACE;IACA;IACA;;EAEF,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU;EACV;EACA;EACA,UAAU;EACV;EACA;;cAUW,2CAA2C;mBACrC;EAEX,KAAK,YAAY,qBAAqB;EAItC,IAAI,aAAa,QAAQ;EAKzB,KAAK,SAAQ,2BAAgC,QAAQ;EAMrD,cAAc,YAAY,SAAS,kBAAkB,QAAQ;EAa7D,YACJ,YACA,OAAO,eACP,qBACC,QAAQ;;cAsBA,6CAA6C;mBACvC;mBACA;UACT;EAER,YAAY;IAAW;;EAIjB,KAAK,YAAY,qBAAqB;EAMtC,IAAI,aAAa,QAAQ;EAKzB,KAAK,SAAQ,2BAAgC,QAAQ;EAKrD,cAAc,YAAY,SAAS,kBAAkB,QAAQ;EAO7D,YACJ,YACA,OAAO,eACP,qBACC,QAAQ;UAOG;UAWA;;iBA8BA,yBAAyB;EACvC;EACA;EACA;EACA,MAAM;EACN,WAAW;EACX,SAAS;EACT,UAAU;EACV,QAAQ;EACR,OAAO;EACP;EACA,WAAW;IACT;;iBAqBY,+BACd,KAAK,kBACL,SAAS,mCACR;;iBAsFa,2BACd,KAAK,kBACL;EAAW;IACV;iBA4Da,oBACd,YAAY,KAAK,iEACjB,SAAQ,sBACP;iBAca,0BACd,YAAY,oBACZ,SAAS,sBACR;iBAuBa,uCACd,cAAc,uBACb;iBAIa,iCACd,YAAY,qBACX;iBA6Ca,oCACd,cAAc,uBACb;iBAoDa,0BACd,cAAc,sBACd;EAAW;IACV;iBA2Ba,+BAA+B,SAAS;iBA+BxC,+BAA+B,QAAQ,SAAS,eAC9D,KAAK,iBAAiB,QAAQ,SAAS,gBACvC;EACE;EACA;EACA,eAAe;EACf,oBAAoB,MAAM,YAAY,QAAQ,SAAS;EACvD,0BACE,MAAM,YAAY,QAAQ,SAAS,mBAChC;EACL;IAED"}
|
package/dist/fuzz.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { i as CostLedger, o as CostReceiptCaptureError } from "./cost-ledger-B1qx30B4.js";
|
|
3
|
-
import { r as varianceBasedCurriculum } from "./active-curriculum-
|
|
3
|
+
import { r as varianceBasedCurriculum } from "./active-curriculum-CD5TU2yW.js";
|
|
4
4
|
//#region src/fuzz/cube.ts
|
|
5
5
|
/** Enumerate every input cell (cartesian product of the axes), in stable order. */
|
|
6
6
|
function enumerateCells(space) {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { a as assertMinted, c as assertRolloutLine, d as isTrainableSplit, h as GATE_POLICIES, m as GATE_CHECK_IDS, n as ROLLOUT_ROLES, r as ROLLOUT_SCHEMA, v as undeclaredStepPayload } from "./schema-C1aaAxTf.js";
|
|
2
|
+
import { a as toRftItems, c as toVerifiersRolloutOutputs, n as toJsonl, o as toSftRows } from "./exporters-Df7TgHFv.js";
|
|
3
3
|
import { basename, dirname, join } from "node:path";
|
|
4
4
|
import { spawnSync } from "node:child_process";
|
|
5
5
|
import { appendFile, mkdir, readFile, writeFile } from "node:fs/promises";
|
|
@@ -758,6 +758,6 @@ async function runRolloutReleaseCli(argv) {
|
|
|
758
758
|
return 0;
|
|
759
759
|
}
|
|
760
760
|
//#endregion
|
|
761
|
-
export {
|
|
761
|
+
export { writeRolloutLedger as S, measureFormatGate as _, addScrubCounts as a, readRolloutJournal as b, scrubLines as c, FORMAT_FILES as d, RELEASE_FORMATS as f, gatedRolloutIds as g, assertGateReport as h, runRolloutReleaseCli as i, scrubRolloutLine as l, FORMAT_GATE_DISPOSITION as m, parseRolloutReleaseArgs as n, defaultRolloutScrubber as o, buildDatasetCard as p, planPushCommand as r, emptyScrubCounts as s, buildHfDataset as t, scrubText as u, releaseRowRefs as v, readRolloutLedger as x, appendRolloutLines as y };
|
|
762
762
|
|
|
763
|
-
//# sourceMappingURL=hf-dataset-
|
|
763
|
+
//# sourceMappingURL=hf-dataset-D8_RNIis.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"hf-dataset-XggBupCr.js","names":[],"sources":["../src/rollout/ledger.ts","../src/rollout/release/gate-report.ts","../src/rollout/release/card.ts","../src/rollout/release/scrub.ts","../src/rollout/release/hf-dataset.ts"],"sourcesContent":["/**\n * Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`\n * lines. Writes validate BEFORE touching disk (a bad line never lands);\n * reads validate line-by-line and fail loud with the line number, because a\n * silently-skipped rollout is a corrupted dataset.\n *\n * \"Validate\" includes the anti-Goodhart invariant (a realness-gated line may\n * not carry a positive reward), so a poisoned line can neither enter a ledger\n * nor leave one.\n *\n * Two read modes, matching the two write-side row classes: `readRolloutLedger`\n * re-validates under the mint policy (training data), `readRolloutJournal`\n * under the write policy (supervision journals, whose unscreened positive\n * rewards are writable and must stay readable).\n */\n\nimport { appendFile, mkdir, readFile, writeFile } from 'node:fs/promises'\nimport { dirname } from 'node:path'\nimport { assertMinted, assertRolloutLine, type MintedRolloutLine, type RolloutLine } from './schema'\n\nfunction serialize(lines: RolloutLine[]): string {\n for (const [i, line] of lines.entries()) assertRolloutLine(line, `rollout line [${i}]`)\n return lines.map((line) => JSON.stringify(line)).join('\\n') + (lines.length > 0 ? '\\n' : '')\n}\n\n/** Replace the ledger file with exactly `lines`. */\nexport async function writeRolloutLedger(path: string, lines: RolloutLine[]): Promise<void> {\n const payload = serialize(lines)\n await mkdir(dirname(path), { recursive: true })\n await writeFile(path, payload)\n}\n\n/** Append `lines` to the ledger file (created if absent). */\nexport async function appendRolloutLines(path: string, lines: RolloutLine[]): Promise<void> {\n if (lines.length === 0) return\n const payload = serialize(lines)\n await mkdir(dirname(path), { recursive: true })\n await appendFile(path, payload)\n}\n\n/**\n * Read and validate every line. Throws on the first malformed/invalid line\n * (with its 1-based line number) — fail-closed, never a silent drop.\n *\n * Validation includes the anti-Goodhart invariant, which is why the result is\n * `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this\n * process from outside the type system (another run, another machine, a\n * hand-edited JSONL), so this read is the runtime boundary where a poisoned\n * line is refused rather than exported.\n */\nexport async function readRolloutLedger(path: string): Promise<MintedRolloutLine[]> {\n return readLines(path, (parsed, context) => assertMinted(parsed, context))\n}\n\n/**\n * Read a ledger under the WRITE-side policy (`validateRolloutLine`), which\n * omits the unscreened-reward check. `writeRolloutLedger` accepts a\n * supervision-journal row (`realness_screened: false` with a positive reward\n * — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says\n * such rows \"must stay writable, readable and reportable\"; a read API that\n * only re-validated under `assertMinted` made every such file unreadable —\n * write-accepted but read-refused is a data-loss trap.\n *\n * The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here\n * can reach a training exporter without passing `assertMinted`, so the\n * promotion gate (which DOES enforce unscreened-reward) is exactly as closed\n * as before. Use `readRolloutLedger` when the file is training data.\n */\nexport async function readRolloutJournal(path: string): Promise<RolloutLine[]> {\n return readLines(path, (parsed, context): RolloutLine => {\n assertRolloutLine(parsed, context)\n return parsed\n })\n}\n\nasync function readLines<T>(\n path: string,\n admit: (parsed: unknown, context: string) => T,\n): Promise<T[]> {\n const raw = await readFile(path, 'utf8')\n const lines: T[] = []\n const rawLines = raw.split('\\n')\n for (let i = 0; i < rawLines.length; i++) {\n const text = rawLines[i]\n if (!text?.trim()) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(text)\n } catch (error) {\n throw new Error(\n `${path}:${i + 1}: malformed JSON — ${error instanceof Error ? error.message : String(error)}`,\n )\n }\n lines.push(admit(parsed, `${path}:${i + 1}`))\n }\n return lines\n}\n","/**\n * Per-format accounting of the anti-Goodhart gate for one dataset release.\n *\n * The defect this exists to make impossible: a dataset card that STATES what\n * the gate does while the build does something else. A sentence in a README is\n * a claim about bytes it never reads, so it drifts the moment an exporter\n * changes — and the drift ships to whoever downloads the dataset.\n *\n * So the card is not allowed to assert anything about the gate. The build\n * measures the rows it is ABOUT TO WRITE (`measureFormatGate`), the measurement\n * is checked against the declared per-format disposition (`assertGateReport`,\n * which throws rather than warns), and the card renders only numbers handed to\n * it. A card that disagrees with its own data files cannot be produced without\n * failing the build first.\n *\n * The dispositions themselves are the release policy, stated once as data:\n *\n * sft EXCLUDE — an SFT row is an imitation target. A gamed\n * trajectory must never be imitated, at any weight.\n * verifiers ZERO_AND_FLAG — reward is a signed learning signal here, so a\n * gamed trajectory at reward 0 is a correct\n * negative. Dropping it would bias the negative\n * population toward honest failures and leave a\n * trainer no example of what gaming looks like\n * when it is penalized.\n * rft ZERO_AND_FLAG — RFT re-samples the completion; only the prompt\n * and the grader's `reference.*` verdict ship, so\n * nothing gamed is imitated. The flag is what lets\n * a grader author skip the instance.\n * raw ZERO_AND_FLAG — a faithful audit dump. Removing rows from it\n * would defeat its only purpose, and the gated\n * row is the one an auditor most wants.\n *\n * `ZERO_AND_FLAG` is never `reward: 0` alone. Zeroing without the label makes a\n * faked success indistinguishable from an honest failure — it hides the gamed\n * population from the buyer instead of disclosing it. Every included format\n * carries `realness_gated` on the row itself.\n *\n * And `ZERO_AND_FLAG` means the whole outcome, not the scalar. A gated row that\n * ships `reward: 0` beside the per-layer verifier scores the reward was\n * computed from has not been zeroed in any sense a trainer respects; the\n * accounting therefore measures every reward-derived number each format writes,\n * not just the one field.\n */\n\nimport type { RftItem, SftRow, VerifiersRolloutOutput } from '../exporters'\nimport {\n GATE_CHECK_IDS,\n GATE_POLICIES,\n type GateCheckId,\n undeclaredStepPayload,\n} from '../gate-checks'\nimport type { MintedRolloutLine } from '../schema'\nimport type { ReleaseFormat } from './card'\n\n/** What a format does with a line the realness gate flagged. */\nexport type GateDisposition = 'exclude' | 'zero-and-flag'\n\nexport const FORMAT_GATE_DISPOSITION: Record<ReleaseFormat, GateDisposition> = {\n sft: 'exclude',\n verifiers: 'zero-and-flag',\n rft: 'zero-and-flag',\n raw: 'zero-and-flag',\n}\n\n/** What the gate accounting reads off an emitted row, per format. */\nexport interface ReleaseRowRef {\n rollout_id: string\n reward: number | null\n /**\n * The rest of the row that was DERIVED from the reward — the per-layer score\n * dict, the judge verdict record, whatever this format ships beside the\n * scalar. Walked for positive numbers, so the certification is about the\n * whole outcome rather than one field.\n *\n * Absent when the format's row carries nothing but the scalar. NOT the whole\n * row: `cost.tokens_in`, `wall_s` and `total_steps` are positive numbers that\n * have nothing to do with the reward, and a certification that flags them is\n * a certification nobody can act on.\n */\n evidence?: unknown\n /**\n * The screen claim AS EMITTED — read off the row, not off the line it came\n * from, because what ships is what matters. Required, not optional: an\n * optional field is how a format quietly opts out of the check that reads it,\n * and every emitted row shape carries `RealnessLabels` precisely so no adapter\n * has to.\n */\n realness_screened: boolean | null\n /**\n * The part of an emitted `steps[]` the wire format does not declare.\n *\n * Separate from `evidence` because the declared step fields are FULL of\n * legitimate positive numbers — `durationMs`, `llm_call_count`,\n * `prompt_token_ids` — and a certification that flags those is one nobody can\n * act on. Only the undeclared remainder is unclassified reward-bearing\n * payload, which is the same partition the check applies.\n *\n * Set only by formats whose row carries steps: today `raw` alone.\n */\n stepEvidence?: unknown\n}\n\n/** A positive number found inside an emitted gated row, with where it was. */\nexport interface EmittedEvidence {\n /** JSON-ish path from the row's evidence root, e.g. `metrics['layer.tests']`. */\n path: string\n value: number\n}\n\nexport interface FormatGateCounts {\n /** Gated lines that reached this format's exporter. */\n input: number\n /** Gated rows the format actually wrote. */\n emitted: number\n /**\n * Gated lines this format did not write. Not all of these are the gate:\n * `verifiers` also drops gap lines (empty transcript) and `rft` drops lines\n * with no prompt turn, so an excluded count can mix both causes.\n */\n excluded: number\n /** Highest reward on an emitted gated row; `null` when none was emitted. */\n maxEmittedReward: number | null\n /**\n * The largest positive number found in the reward-DERIVED payload of an\n * emitted gated row, and its path; `null` when there is none.\n *\n * This column exists because the release once certified CLEAN while leaking.\n * `assertGateReport` inspected `outcome.reward` alone, so a gated row shipping\n * `reward: 0` next to `metrics['layer.tests']: 1` — the deterministic verifier\n * score the reward was computed from, and the per-rubric score dict of the\n * Prime Intellect verifiers format — passed, and the card rendered \"max reward\n * | 0\" over a file that carried the gamed signal at full value. A wrong\n * certification is worse than the leak: it is the leak plus a document saying\n * there isn't one.\n */\n maxEmittedEvidence: EmittedEvidence | null\n /**\n * Rows this format wrote carrying a positive reward whose producer DECLARED\n * that no authenticity screen ever ran on it (`realness_screened: false`).\n *\n * Measured over EVERY emitted row, not just the gated ones: an unscreened\n * reward is by definition one the gate never had a verdict on, so it is not in\n * the gated set and a measurement scoped to that set would report 0 forever.\n * `assertMinted` already refuses these, which is exactly why the release still\n * measures them — the last door before a public dataset does not get to assume\n * the earlier doors held.\n */\n unscreenedPositiveRows: number\n /** Highest reward on such a row; `null` when there is none. */\n maxUnscreenedReward: number | null\n /**\n * The largest positive number found in an emitted gated row's UNDECLARED\n * per-step payload, and its path; `null` when there is none.\n *\n * The column exists because the gate read `outcome` and nothing else for\n * three rounds, so a gated line shipping `steps: [{kind, name, reward: 0.86}]`\n * certified clean — the release accounting agreed with the exporter that a\n * per-step reward was not a reward.\n */\n maxEmittedStepEvidence: EmittedEvidence | null\n}\n\nexport interface GateReport {\n /** Gated lines in the release input, after the split/proposer filters. */\n gatedLines: number\n byFormat: Partial<Record<ReleaseFormat, FormatGateCounts>>\n}\n\n/** Rollout ids of every gated line, the key the emitted rows are matched on. */\nexport function gatedRolloutIds(lines: readonly MintedRolloutLine[]): Set<string> {\n return new Set(lines.filter((line) => line.outcome.realness_gated).map((line) => line.rollout_id))\n}\n\n/**\n * Row refs per format. Written as one adapter per format so that the knowledge\n * of WHERE the id and reward live in each published shape sits next to the\n * assertion that uses it — an exporter that moves either field breaks here\n * rather than silently reporting zero gated rows.\n */\nexport const releaseRowRefs = {\n // An SFT row is `{messages, metadata}` and `metadata` holds ids plus the\n // scalar — no reward-derived payload beyond `reward` itself, and the gated\n // disposition is EXCLUDE anyway.\n sft: (rows: readonly SftRow[]): ReleaseRowRef[] =>\n rows.map((row) => ({\n rollout_id: row.metadata.rollout_id,\n reward: row.metadata.reward,\n realness_screened: row.metadata.realness_screened,\n })),\n // `metrics` IS the per-rubric score dict in the Prime Intellect verifiers\n // format — the same numbers the reward was computed from. This is the leak\n // that shipped.\n verifiers: (rows: readonly VerifiersRolloutOutput[]): ReleaseRowRef[] =>\n rows.map((row) => ({\n rollout_id: row.info.rollout_id,\n reward: row.reward,\n evidence: row.metrics,\n realness_screened: row.info.realness_screened,\n })),\n // RFT re-samples the completion, but the grader reads `item.reference.*`, so\n // the verbatim judge verdict is a reward-bearing field on the row.\n rft: (rows: readonly RftItem[]): ReleaseRowRef[] =>\n rows.map((row) => ({\n rollout_id: row.reference.rollout_id,\n reward: row.reference.reward,\n evidence: row.reference.verdict,\n realness_screened: row.reference.realness_screened,\n })),\n raw: (lines: readonly MintedRolloutLine[]): ReleaseRowRef[] =>\n lines.map((line) => ({\n rollout_id: line.rollout_id,\n reward: line.outcome.reward,\n // `provenance.gated_evidence` is deliberately NOT walked: relocating the\n // diagnostics there is what the gate DOES, and the raw config is the\n // audit dump those diagnostics exist for.\n evidence: { metrics: line.outcome.metrics, verdict: line.outcome.verdict },\n // `raw` is the only config that writes the whole line, so it is the only\n // one whose rows can carry a per-step reward.\n stepEvidence: undeclaredStepPayload(line.steps),\n realness_screened: line.outcome.realness_screened ?? null,\n })),\n}\n\n/** Every positive finite number inside a row's reward-derived payload, with its path. */\nfunction positiveNumbersIn(value: unknown, path: string): EmittedEvidence[] {\n if (typeof value === 'number') {\n return Number.isFinite(value) && value > 0 ? [{ path, value }] : []\n }\n if (Array.isArray(value)) {\n return value.flatMap((item, i) => positiveNumbersIn(item, `${path}[${i}]`))\n }\n if (typeof value === 'object' && value !== null) {\n return Object.entries(value).flatMap(([key, child]) =>\n positiveNumbersIn(child, path === '' ? key : `${path}.${key}`),\n )\n }\n return []\n}\n\n/** Measure one format's gated rows from the refs of the rows about to be written. */\nexport function measureFormatGate(\n gated: ReadonlySet<string>,\n refs: readonly ReleaseRowRef[],\n): FormatGateCounts {\n const emitted = refs.filter((ref) => gated.has(ref.rollout_id))\n const rewards = emitted.map((ref) => ref.reward).filter((r): r is number => r !== null)\n const evidence = emitted.flatMap((ref) => positiveNumbersIn(ref.evidence, ''))\n const stepEvidence = emitted.flatMap((ref) => positiveNumbersIn(ref.stepEvidence, 'steps'))\n const unscreened = refs\n .filter((ref) => ref.realness_screened === false)\n .map((ref) => ref.reward)\n .filter((r): r is number => r !== null && r > 0)\n return {\n input: gated.size,\n emitted: emitted.length,\n excluded: gated.size - emitted.length,\n maxEmittedReward: rewards.length === 0 ? null : Math.max(...rewards),\n maxEmittedEvidence: evidence.reduce<EmittedEvidence | null>(\n (best, found) => (best === null || found.value > best.value ? found : best),\n null,\n ),\n unscreenedPositiveRows: unscreened.length,\n maxUnscreenedReward: unscreened.length === 0 ? null : Math.max(...unscreened),\n maxEmittedStepEvidence: stepEvidence.reduce<EmittedEvidence | null>(\n (best, found) => (best === null || found.value > best.value ? found : best),\n null,\n ),\n }\n}\n\n/**\n * The measured form of each canonical gate check, over the rows a release is\n * ABOUT TO WRITE. Returns the failure message, or `null` when the format is\n * clean on that check.\n *\n * TOTAL over `GateCheckId` — this map and `GATE_POLICIES.assertGateReport` are\n * the two things a new check breaks here, so the release certifier cannot be\n * left behind by a check added anywhere else in the package. That is the whole\n * point: for four rounds each guard composed its own subset by hand, and a\n * release certifying CLEAN while leaking is the most expensive version of that\n * mistake, because it is the leak plus a document saying there isn't one.\n */\nconst REPORT_MEASURES: {\n readonly [K in GateCheckId]: (format: ReleaseFormat, counts: FormatGateCounts) => string | null\n} = {\n 'reward-relationship': (format, counts) =>\n counts.maxEmittedReward === null || counts.maxEmittedReward <= 0\n ? null\n : `release format \"${format}\": ${counts.emitted} realness-gated row(s) carry a positive ` +\n `reward (max ${counts.maxEmittedReward}). A run flagged as gamed may not ship a ` +\n 'positive reward in any config.',\n 'gated-evidence': (format, counts) => {\n if (counts.maxEmittedEvidence === null) return null\n const { path, value } = counts.maxEmittedEvidence\n return (\n `release format \"${format}\": ${counts.emitted} realness-gated row(s) carry a positive ` +\n `reward-derived number (${path} = ${value}). Zeroing the scalar is not enough — the ` +\n 'per-layer scores and judge verdict a fabricated reward was computed FROM are the ' +\n 'same signal in component form, and in this format they are read as training input. ' +\n 'They belong in `provenance.gated_evidence` (see `gateGamedOutcome`), not on the row.'\n )\n },\n 'undeclared-step-payload': (format, counts) => {\n if (counts.maxEmittedStepEvidence === null) return null\n const { path, value } = counts.maxEmittedStepEvidence\n return (\n `release format \"${format}\": ${counts.emitted} realness-gated row(s) carry a positive ` +\n `per-step number under a field \\`tangle.rollout.v1\\` does not declare (${path} = ${value}). ` +\n 'A per-step reward is the same training signal as the scalar, in credit-assignment form, ' +\n 'and the exporters copy `steps` through verbatim — so the row ships it beside a `reward` ' +\n 'of 0. It belongs in `provenance.gated_evidence.steps` (see `gateGamedOutcome`).'\n )\n },\n 'unscreened-reward': (format, counts) =>\n counts.maxUnscreenedReward === null\n ? null\n : `release format \"${format}\": ${counts.unscreenedPositiveRows} row(s) carry a positive ` +\n `reward (max ${counts.maxUnscreenedReward}) whose producer declares that NO authenticity ` +\n 'screen ran on it (`realness_screened: false`). Nothing has established those successes ' +\n 'are real, and a published dataset may not present an unqualified verdict as a measured ' +\n 'one. Screen the runs, or publish them at `reward: null`.',\n}\n\n/**\n * Fail the build when the measurement disagrees with the declared policy.\n *\n * Throws, never filters: an emitted positive reward on a gated row means an\n * exporter upstream stopped applying the gate, and silently dropping the row\n * would hide the producer that made it — the producer is the actual defect.\n *\n * Certifies the whole emitted outcome, not `reward` alone. The earlier version\n * checked one field and therefore certified a release CLEAN while its\n * `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in\n * the top-level `metrics` dict — the card then rendered \"max reward | 0\" over\n * exactly that file. A certification that is wrong is worse than an\n * uncertified leak, so the checks it runs are no longer written down here at\n * all: it iterates `GATE_CHECK_IDS` under its own declared policy.\n */\nexport function assertGateReport(report: GateReport): void {\n for (const [format, counts] of Object.entries(report.byFormat) as Array<\n [ReleaseFormat, FormatGateCounts]\n >) {\n for (const id of GATE_CHECK_IDS) {\n if (GATE_POLICIES.assertGateReport[id].kind !== 'enforce') continue\n const failure = REPORT_MEASURES[id](format, counts)\n if (failure !== null) throw new Error(failure)\n }\n // Not a gate check: this one is the RELEASE POLICY (`FORMAT_GATE_DISPOSITION`)\n // rather than the anti-Goodhart invariant — it asks whether the format did\n // what it said it would with gated rows, not whether a row is poisoned.\n if (FORMAT_GATE_DISPOSITION[format] === 'exclude' && counts.emitted > 0) {\n throw new Error(\n `release format \"${format}\" declares gated lines EXCLUDED but wrote ${counts.emitted} ` +\n 'of them.',\n )\n }\n }\n}\n","/**\n * HuggingFace dataset-card (README.md) generation for a rollout-ledger release.\n *\n * The card is a pure function of the SCRUBBED lines plus the release options —\n * no timestamps, no environment reads — so rebuilding from the same ledger\n * yields byte-identical output. It documents the schema, provenance (run ids,\n * generations, the official judge), per-role reward semantics including the\n * inherited/contribution caveat, and a role × reward counts table.\n */\n\nimport { type MintedRolloutLine, ROLLOUT_ROLES, ROLLOUT_SCHEMA, type RolloutRole } from '../schema'\nimport { assertGateReport, FORMAT_GATE_DISPOSITION, type GateReport } from './gate-report'\nimport type { ScrubCounts } from './scrub'\n\nexport const RELEASE_FORMATS = ['sft', 'verifiers', 'rft', 'raw'] as const\nexport type ReleaseFormat = (typeof RELEASE_FORMATS)[number]\n\n/** Format → data file path inside the dataset dir (train split only). */\nexport const FORMAT_FILES: Record<ReleaseFormat, string> = {\n sft: 'sft/train.jsonl',\n verifiers: 'verifiers/train.jsonl',\n rft: 'rft/train.jsonl',\n raw: 'raw/train.jsonl',\n}\n\nconst FORMAT_DESCRIPTIONS: Record<ReleaseFormat, string> = {\n sft: 'Successful trainable-split transcripts (`reward >= 1`, never realness-gated) as `{messages, metadata}` chat JSONL.',\n verifiers:\n 'Prime Intellect verifiers `RolloutOutput`: prompt/completion split at the first assistant turn, plus reward, metrics, tool defs, and token usage.',\n rft: 'OpenAI RFT items: prompt turns plus `reference.*` verdict fields for a grader (completions are re-sampled during RFT).',\n raw: `Full \\`${ROLLOUT_SCHEMA}\\` ledger lines (scrubbed), one per agent invocation.`,\n}\n\nexport interface DatasetCardInputs {\n /** Scrubbed, release-filtered lines (what actually ships). */\n lines: MintedRolloutLine[]\n formats: ReleaseFormat[]\n includeProposers: boolean\n /** Source ledger basenames, for provenance. */\n sourceFiles: string[]\n scrubTotals: ScrubCounts\n excluded: { proposers: number; nonTrain: number }\n formatCounts: Partial<Record<ReleaseFormat, number>>\n /**\n * Per-format anti-Goodhart accounting MEASURED on the rows the build wrote.\n * Required, not optional: the card's only statement about the gate is a\n * render of these numbers, so a card cannot be produced without them and\n * cannot drift from the data files it ships beside.\n */\n gate: GateReport\n}\n\nfunction unique(values: Array<string | null>): string[] {\n return [...new Set(values.filter((v): v is string => v !== null))].sort()\n}\n\nfunction formatReward(reward: number | null): string {\n if (reward === null) return 'null'\n return Number.isInteger(reward) ? String(reward) : reward.toFixed(4)\n}\n\nfunction markdownTable(header: string[], rows: string[][]): string {\n return [\n `| ${header.join(' | ')} |`,\n `| ${header.map(() => '---').join(' | ')} |`,\n ...rows.map((row) => `| ${row.join(' | ')} |`),\n ].join('\\n')\n}\n\n/** Where the anti-Goodhart flag lives on each format's emitted row. */\nconst GATE_FLAG_FIELD: Record<ReleaseFormat, string> = {\n sft: '— (no gated row is written)',\n verifiers: '`info.realness_gated`',\n rft: '`reference.realness_gated`',\n raw: '`outcome.realness_gated`',\n}\n\nconst GATE_POLICY_LABEL: Record<ReleaseFormat, string> = {\n sft: 'EXCLUDED — an SFT row is an imitation target',\n verifiers: 'INCLUDED, reward forced to 0, flagged',\n rft: 'INCLUDED, reward forced to 0, flagged',\n raw: 'INCLUDED verbatim, flagged',\n}\n\nfunction gateSection(formats: ReleaseFormat[], gate: GateReport, totalLines: number): string {\n const rows = formats.map((format) => {\n const counts = gate.byFormat[format]\n return [\n format,\n GATE_POLICY_LABEL[format],\n String(counts?.emitted ?? 0),\n String(counts?.excluded ?? 0),\n counts?.maxEmittedReward === null || counts === undefined\n ? '—'\n : formatReward(counts.maxEmittedReward),\n counts?.maxEmittedEvidence == null\n ? '—'\n : `${counts.maxEmittedEvidence.path} = ${counts.maxEmittedEvidence.value}`,\n GATE_FLAG_FIELD[format],\n ]\n })\n const mixedExclusion = formats.some(\n (format) => FORMAT_GATE_DISPOSITION[format] === 'zero-and-flag',\n )\n return [\n `\\`outcome.realness_gated: true\\` marks a run whose success signal was faked (it satisfied the proxy without doing the work). **${gate.gatedLines} of ${totalLines} lines in this release are gated.**`,\n '',\n 'The gate is enforced, not asserted. A line pairing `realness_gated: true` with a reward above 0 is rejected by the schema validator, so it cannot be read out of a source ledger or written into `raw/train.jsonl` at all; on a gated line the numbers the reward was computed from (`outcome.metrics`, the verbatim judge `outcome.verdict`, and any per-step field the schema does not declare — a per-step reward is the same signal in credit-assignment form) are moved out of `outcome` and `steps[]` and into `provenance.gated_evidence`, where no config reads them as training input; the exporters re-check the same invariant on every line; and the release build measures the rows it is about to write and refuses to write ANY file if a gated row carries a positive reward — or any positive number derived from one — in ANY config.',\n '',\n 'The table below is **measured on the rows in this release**, not a description of intent — the build counts the emitted rows and this card renders those counts:',\n '',\n markdownTable(\n [\n 'config',\n 'policy',\n 'gated rows written',\n 'gated rows not written',\n 'max reward',\n 'max reward-derived number',\n 'flag',\n ],\n rows,\n ),\n '',\n 'Why gated rows are kept where they are kept: an SFT row is imitated verbatim, so a gamed trajectory must never appear in one at any weight. In `verifiers` the reward is a signed learning signal, and a gamed trajectory at reward 0 is a correct negative — dropping it would bias the negative population toward honest failures and leave a trainer no example of gaming being penalized. `rft` re-samples the completion, so only the prompt and the grader reference ship. `raw` is a faithful audit dump, where the gated row is the one an auditor most wants.',\n '',\n \"Reward 0 is never the only label: every included config carries the flag on the row itself, because zeroing alone makes a faked success indistinguishable from an honest failure. Filter on the flag to drop the gamed population, or select on it to mine it. The gated run's own measurements are not destroyed either — they are parked verbatim under `provenance.gated_evidence` in `raw/train.jsonl`, which is where an auditor can see what the run claimed and why it was flagged.\",\n mixedExclusion\n ? '\\n\"Gated rows not written\" is not all gate: `verifiers` also drops gap lines (empty transcript) and `rft` drops lines with no prompt turn, so that column can mix both causes.'\n : '',\n ]\n .join('\\n')\n .trimEnd()\n}\n\nfunction roleRewardRows(lines: MintedRolloutLine[]): string[][] {\n const counts = new Map<RolloutRole, Map<string, number>>()\n for (const line of lines) {\n const byReward = counts.get(line.role) ?? new Map<string, number>()\n const key = formatReward(line.outcome.reward)\n byReward.set(key, (byReward.get(key) ?? 0) + 1)\n counts.set(line.role, byReward)\n }\n const rows: string[][] = []\n for (const role of ROLLOUT_ROLES) {\n const byReward = counts.get(role)\n if (!byReward) continue\n const keys = [...byReward.keys()].sort((a, b) => {\n if (a === 'null') return 1\n if (b === 'null') return -1\n return Number(b) - Number(a)\n })\n for (const key of keys) rows.push([role, key, String(byReward.get(key))])\n }\n return rows\n}\n\nexport function buildDatasetCard(inputs: DatasetCardInputs): string {\n const {\n lines,\n formats,\n includeProposers,\n sourceFiles,\n scrubTotals,\n excluded,\n formatCounts,\n gate,\n } = inputs\n // The card is the artifact a buyer reads; it may not render a gate report\n // that violates the release policy even when called outside the build.\n assertGateReport(gate)\n\n const runIds = unique(lines.map((line) => line.run_id))\n const generations = [...new Set(lines.map((line) => line.generation))]\n .filter((g): g is number => g !== null)\n .sort((a, b) => a - b)\n const models = unique(lines.map((line) => line.policy.model))\n const harnesses = unique(lines.map((line) => line.policy.harness))\n const captures = unique(lines.map((line) => line.provenance.capture))\n const rewardSources = unique(lines.map((line) => line.outcome.reward_source))\n const gapLines = lines.filter((line) => line.messages.length === 0).length\n const gatedLines = lines.filter((line) => line.outcome.realness_gated).length\n if (gatedLines !== gate.gatedLines) {\n throw new Error(\n `dataset card: gate report claims ${gate.gatedLines} realness-gated line(s) but the lines ` +\n `it describes contain ${gatedLines} — the card and the build measured different data.`,\n )\n }\n\n const configs = formats\n .map((format) =>\n [\n ` - config_name: ${format}`,\n ' data_files:',\n ' - split: train',\n ` path: ${FORMAT_FILES[format]}`,\n ].join('\\n'),\n )\n .join('\\n')\n\n const frontmatter = [\n '---',\n 'license: unknown',\n 'pretty_name: Tangle rollout ledger — agent trajectories',\n 'configs:',\n configs,\n '---',\n ].join('\\n')\n\n const formatsTable = markdownTable(\n ['config', 'path', 'rows', 'contents'],\n formats.map((format) => [\n format,\n `\\`${FORMAT_FILES[format]}\\``,\n String(formatCounts[format] ?? 0),\n FORMAT_DESCRIPTIONS[format],\n ]),\n )\n\n const countsTable = markdownTable(['role', 'reward', 'lines'], roleRewardRows(lines))\n\n const scrubTable = markdownTable(\n ['rule', 'rewrites'],\n Object.entries(scrubTotals).map(([rule, count]) => [rule, String(count)]),\n )\n\n const proposerNote = includeProposers\n ? 'Proposer sessions are INCLUDED (`--include-proposers`); their transcripts contain improvement-loop harness source.'\n : `Proposer sessions are excluded by default (${excluded.proposers} lines dropped); they contain improvement-loop harness source. Rebuild with \\`--include-proposers\\` to keep them.`\n\n return `${frontmatter}\n\n# Tangle rollout ledger — agent trajectories\n\nOne line per agent invocation (supervisor episode, worker session, proposer shot, judge call, analyst pass) captured by the \\`${ROLLOUT_SCHEMA}\\` rollout ledger, labeled with improvement-loop coordinates and the official-judge reward, with the full message transcript inline.\n\nThis release contains the **trainable split only** (\\`search\\`). Holdout, dev, and canary splits are structurally excluded at build time, and the build additionally drops any non-trainable line as a fail-closed filter (${excluded.nonTrain} dropped here).\n\n## Formats\n\n${formatsTable}\n\n## Anti-Goodhart gate (\\`realness_gated\\`)\n\n${gateSection(formats, gate, lines.length)}\n\n## Schema (\\`${ROLLOUT_SCHEMA}\\`)\n\nEach raw line carries:\n\n- \\`rollout_id\\` / \\`parent_rollout_id\\` — invocation identity; workers point at their spawning supervisor episode.\n- \\`run_id\\`, \\`experiment_id\\`, \\`candidate_id\\` — run/experiment/candidate identity from the producing RunRecord, when present.\n- \\`generation\\`, \\`candidate_index\\` — improvement-loop coordinates (\\`-1\\` = baseline campaign; \\`null\\` = not an improvement loop).\n- \\`role\\` — one of ${ROLLOUT_ROLES.map((role) => `\\`${role}\\``).join(', ')}.\n- \\`task\\` — suite, instance id, split, seed, replicate index.\n- \\`policy\\` — harness, model, provider, profile commit, prompt/config hashes, sampling params.\n- \\`messages\\` / \\`tool_defs\\` — full transcript in canonical OpenAI chat-with-tools form (including \\`reasoning_content\\`). An empty \\`messages\\` array is a labeled gap line; \\`provenance.gap\\` says why the transcript could not be recovered (${gapLines} gap lines in this release).\n- \\`outcome\\` — \\`reward\\` (the single scalar), \\`reward_source\\`, the verbatim judge \\`verdict\\`, non-scalar \\`metrics\\`, and \\`realness_gated\\` (anti-Goodhart flag; see [Anti-Goodhart gate](#anti-goodhart-gate-realness_gated) for the per-config treatment and the measured counts).\n- \\`cost\\` — usd, token counts, wall time.\n- \\`artifacts\\` / \\`provenance\\` — patch/run-dir/transcript pointers (scrubbed) and capture metadata.\n\n## Provenance\n\n- Source ledgers: ${sourceFiles.map((file) => `\\`${file}\\``).join(', ')}\n- Run ids: ${runIds.map((id) => `\\`${id}\\``).join(', ')}\n- Generations: ${generations.join(', ')} (\\`-1\\` = baseline campaign)\n- Models: ${models.map((m) => `\\`${m}\\``).join(', ')}\n- Harnesses: ${harnesses.map((h) => `\\`${h}\\``).join(', ')}\n- Capture modes: ${captures.join(', ')}\n- Every reward traces to a named source (reward sources in this release: ${rewardSources.map((s) => `\\`${s}\\``).join(', ')}).\n\n${proposerNote}\n\n## Reward semantics per role\n\n- **agent** — the producing RunRecord's holdout/search score, with the realness gate forcing gamed successes to 0.\n- **supervisor** — the official-judge verdict on the episode's delivered artifact (1 = resolved, 0 = not).\n- **worker** — INHERITED from the parent supervisor episode (\\`…/inherited\\`). Caveat: reward 1 does not establish this worker's individual contribution (sibling workers in the same episode share the episode outcome), and reward 0 does not prove this worker failed.\n- **proposer** — the fraction of improvement-set instances the proposed candidate resolved (\\`…/candidate-resolved-fraction\\`); a scalar in [0, 1], not a binary verdict.\n- **judge / analyst** — carry the episode verdict where one applies; otherwise \\`reward: null\\` (a labeled gap, never 0).\n\n## Counts\n\n${countsTable}\n\nTotal lines: ${lines.length}\n\n## Scrubbing\n\nAbsolute home paths were rewritten to \\`$WORK\\`, credential-shaped strings were replaced with \\`[REDACTED:<kind>]\\` markers, internal hostnames were normalized to \\`*.internal.example\\`, and username-bearing incidentals (\\`ls -l\\` owner columns, per-user pytest tmpdirs) were normalized to \\`user\\` / \\`$USER\\`. Rewrite counts for this release (full per-file breakdown in \\`scrub-report.json\\`):\n\n${scrubTable}\n\n## License\n\n\\`license: unknown\\` is a placeholder — the releasing operator must set the real SPDX license id in the frontmatter above before publishing.\n\n## Citation\n\n\\`\\`\\`bibtex\n@misc{tangle_rollout_ledger,\n title = {Tangle rollout ledger — agent trajectories},\n author = {{Tangle Network}},\n howpublished = {HuggingFace Datasets},\n note = {Operator: fill in the repository URL, authors, and year before publishing}\n}\n\\`\\`\\`\n`\n}\n","/**\n * Deterministic scrubbing pass over rollout-ledger lines before public release.\n *\n * Every rule is a pure regex rewrite applied to every string value in a line\n * (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed\n * line is still a valid `tangle.rollout.v1` line. Rules are idempotent:\n * scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is\n * the property the release pipeline relies on to prove nothing half-scrubbed\n * ships. Rule order matters: whole `KEY=value` env pairs are redacted before\n * the bare-key rule so one secret is never counted twice.\n */\n\nimport { assertMinted, type MintedRolloutLine } from '../schema'\n\nexport interface ScrubRule {\n name: string\n pattern: RegExp\n /** Rewrite for one match; `g1` is the first capture group when present. */\n rewrite: (match: string, g1?: string) => string\n}\n\nexport const SCRUB_RULES: readonly ScrubRule[] = [\n {\n // /home/<user> and /Users/<user> prefixes → $WORK ($HOME-derived paths).\n name: 'home-path',\n pattern: /\\/(?:home|Users)\\/[A-Za-z0-9._-]+/g,\n rewrite: () => '$WORK',\n },\n {\n // Harness stores encode cwd as a dashed path segment (-home-drew-…);\n // strip the leading -home-<user> so the username never ships.\n name: 'home-path-encoded',\n pattern: /(?<=\\/)-(?:home|Users)-[A-Za-z0-9_.]+/g,\n rewrite: () => '$WORK',\n },\n {\n // pytest names its tmpdir after the invoking user (/tmp/pytest-of-drew).\n name: 'tmp-user-dir',\n pattern: /\\/tmp\\/pytest-of-[A-Za-z0-9._-]+/g,\n rewrite: () => '/tmp/pytest-of-$USER',\n },\n {\n // `ls -l` owner/group columns leak the username in captured tool output.\n // The `(?!user user…)` guard keeps a second pass at zero rewrites.\n name: 'ls-owner',\n pattern:\n /([-bcdlps][-rwxsStT]{9}[.+@]?\\s+\\d+\\s+)(?!user user(?=\\s))[A-Za-z0-9._-]+\\s+[A-Za-z0-9._-]+(?=\\s)/g,\n rewrite: (_match, prefix) => `${prefix}user user`,\n },\n {\n // Env-var-shaped secrets: NAME=value where NAME looks credential-bearing.\n // The (?!\\[REDACTED:) guard keeps the rule idempotent on its own output.\n name: 'env-secret',\n pattern:\n /\\b([A-Z][A-Z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIALS?))=(?!\\[REDACTED:)(\"[^\"]*\"|'[^']*'|[^\\s\"']+)/g,\n rewrite: (_match, name) => `${name}=[REDACTED:env]`,\n },\n {\n name: 'bearer-token',\n pattern: /\\b(Bearer|Basic)\\s+(?!\\[REDACTED:)[A-Za-z0-9\\-._~+/=]{8,}/g,\n rewrite: (_match, scheme) => `${scheme} [REDACTED:bearer]`,\n },\n {\n // Bare provider-prefixed keys (OpenAI/Anthropic sk-, GitHub ghp_/gho_/…,\n // fine-grained PATs, HuggingFace hf_, Slack xox*, AWS AKIA).\n name: 'api-key',\n pattern:\n /\\b(?:sk-[A-Za-z0-9_-]{16,}|gh[pousr]_[A-Za-z0-9]{16,}|github_pat_[A-Za-z0-9_]{20,}|hf_[A-Za-z0-9]{16,}|xox[baprs]-[A-Za-z0-9-]{10,}|AKIA[0-9A-Z]{16})\\b/g,\n rewrite: () => '[REDACTED:api-key]',\n },\n {\n // Our infra hostnames → placeholder domain, subdomain preserved\n // (router.tangle.tools → router.internal.example).\n name: 'infra-host',\n pattern: /(?<![A-Za-z0-9.-])((?:[A-Za-z0-9-]+\\.)*)tangle\\.(?:tools|network)(?![A-Za-z0-9-])/g,\n rewrite: (_match, prefix) => `${prefix ?? ''}internal.example`,\n },\n {\n // Known workstation hostnames leak through email Message-IDs and fqdn\n // lookups inside captured test output; extend the list as machines join.\n name: 'machine-host',\n pattern: /\\b[A-Za-z0-9]+-GTR-Pro\\b/g,\n rewrite: () => 'workstation',\n },\n]\n\n/** Rule name → number of matches rewritten. Always carries every rule (0 is data). */\nexport type ScrubCounts = Record<string, number>\n\nexport function emptyScrubCounts(): ScrubCounts {\n return Object.fromEntries(SCRUB_RULES.map((rule) => [rule.name, 0]))\n}\n\nexport function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts {\n for (const [name, count] of Object.entries(from)) into[name] = (into[name] ?? 0) + count\n return into\n}\n\nexport function scrubText(text: string, counts: ScrubCounts): string {\n let out = text\n for (const rule of SCRUB_RULES) {\n out = out.replace(rule.pattern, (match: string, ...rest: unknown[]) => {\n counts[rule.name] = (counts[rule.name] ?? 0) + 1\n return rule.rewrite(match, typeof rest[0] === 'string' ? rest[0] : undefined)\n })\n }\n return out\n}\n\nfunction scrubValue(value: unknown, counts: ScrubCounts): unknown {\n if (typeof value === 'string') return scrubText(value, counts)\n if (Array.isArray(value)) return value.map((item) => scrubValue(item, counts))\n if (value !== null && typeof value === 'object') {\n const out: Record<string, unknown> = {}\n for (const [key, item] of Object.entries(value)) out[key] = scrubValue(item, counts)\n return out\n }\n return value\n}\n\n/**\n * Scrub every string value in a line; structure and key order are preserved.\n *\n * `assertMinted` on the way out rather than a cast: scrubbing rebuilds the\n * object, so the brand has to be re-earned, and re-validating proves the rules\n * did not rewrite a field the schema constrains (`reward` is a number, not a\n * string, so no rule should ever touch it — this is what checks that).\n */\nexport function scrubRolloutLine(line: MintedRolloutLine, counts: ScrubCounts): MintedRolloutLine {\n return assertMinted(scrubValue(line, counts), `scrubbed rollout line ${line.rollout_id}`)\n}\n\nexport function scrubLines(lines: MintedRolloutLine[]): {\n lines: MintedRolloutLine[]\n counts: ScrubCounts\n} {\n const counts = emptyScrubCounts()\n return { lines: lines.map((line) => scrubRolloutLine(line, counts)), counts }\n}\n\n/**\n * A `RolloutScrubber` (text → text) applying the full rule set — the\n * default hook to pass to `mintRolloutRows({ scrub })` so lines are\n * scrubbed at mint time, before they ever reach a ledger file. Release\n * builds re-run `scrubLines` regardless (idempotent), so double-scrubbing\n * is safe and counted as zero.\n */\nexport function defaultRolloutScrubber(text: string): string {\n return scrubText(text, emptyScrubCounts())\n}\n","/**\n * One-command HuggingFace dataset release from rollout ledgers:\n *\n * agent-eval rollout-release <ledger.jsonl...> --out <dir> \\\n * [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]\n *\n * Pipeline per input ledger: read + validate → fail-closed filters\n * (trainable split only; proposer sessions dropped unless\n * --include-proposers, they contain improvement-loop harness source) →\n * deterministic scrub → export the requested formats + scrub-report.json +\n * auto-generated README.md card. Deterministic: same inputs and flags →\n * byte-identical output dir.\n *\n * --push uploads the built dir with `huggingface-cli upload` only when the\n * CLI exists on PATH and HF_TOKEN is present in the env; the token is\n * never printed. Everything else runs fully offline.\n */\n\nimport { spawnSync } from 'node:child_process'\nimport { mkdir, writeFile } from 'node:fs/promises'\nimport { basename, dirname, join } from 'node:path'\nimport { toJsonl, toRftItems, toSftRows, toVerifiersRolloutOutputs } from '../exporters'\nimport { readRolloutLedger, writeRolloutLedger } from '../ledger'\nimport { isTrainableSplit, type MintedRolloutLine } from '../schema'\nimport { buildDatasetCard, FORMAT_FILES, RELEASE_FORMATS, type ReleaseFormat } from './card'\nimport {\n assertGateReport,\n FORMAT_GATE_DISPOSITION,\n type GateReport,\n gatedRolloutIds,\n measureFormatGate,\n releaseRowRefs,\n} from './gate-report'\nimport { addScrubCounts, emptyScrubCounts, type ScrubCounts, scrubLines } from './scrub'\n\nexport interface BuildOptions {\n out: string\n formats: ReleaseFormat[]\n includeProposers: boolean\n}\n\nexport interface ScrubReport {\n /** Input ledger path → rule → rewrite count (only shipped lines are scrubbed). */\n files: Record<string, ScrubCounts>\n totals: ScrubCounts\n excluded: { proposers: number; nonTrain: number }\n}\n\nexport interface BuildSummary {\n inputs: string[]\n read: number\n kept: number\n scrub: ScrubReport\n formatCounts: Partial<Record<ReleaseFormat, number>>\n /** Per-format anti-Goodhart accounting, measured on the rows written. */\n gate: GateReport\n files: string[]\n}\n\nexport async function buildHfDataset(\n inputs: string[],\n options: BuildOptions,\n): Promise<BuildSummary> {\n if (inputs.length === 0) throw new Error('no input ledgers given')\n if (options.formats.length === 0) throw new Error('no formats selected')\n\n const report: ScrubReport = {\n files: {},\n totals: emptyScrubCounts(),\n excluded: { proposers: 0, nonTrain: 0 },\n }\n const kept: MintedRolloutLine[] = []\n let read = 0\n\n for (const input of inputs) {\n const lines = await readRolloutLedger(input)\n read += lines.length\n const shippable = lines.filter((line) => {\n if (!isTrainableSplit(line.task.split)) {\n report.excluded.nonTrain += 1\n return false\n }\n if (!options.includeProposers && line.role === 'proposer') {\n report.excluded.proposers += 1\n return false\n }\n return true\n })\n const scrubbed = scrubLines(shippable)\n report.files[input] = scrubbed.counts\n addScrubCounts(report.totals, scrubbed.counts)\n kept.push(...scrubbed.lines)\n }\n\n const formatCounts: Partial<Record<ReleaseFormat, number>> = {}\n const files: string[] = []\n const gated = gatedRolloutIds(kept)\n const gate: GateReport = { gatedLines: gated.size, byFormat: {} }\n\n // Every selected format's rows are exported and gate-measured BEFORE the\n // first byte is written. A build that would ship a gamed run at a positive\n // reward fails with nothing on disk, rather than leaving a poisoned config\n // behind for someone to `--push`.\n const pending: Array<{ path: string; write: () => Promise<void> }> = []\n\n for (const format of options.formats) {\n const path = join(options.out, FORMAT_FILES[format])\n if (format === 'raw') {\n // writeRolloutLedger re-validates every scrubbed line before it lands.\n gate.byFormat.raw = measureFormatGate(gated, releaseRowRefs.raw(kept))\n formatCounts.raw = kept.length\n pending.push({ path, write: () => writeRolloutLedger(path, kept) })\n } else if (format === 'sft') {\n const rows = toSftRows(kept)\n gate.byFormat.sft = measureFormatGate(gated, releaseRowRefs.sft(rows))\n formatCounts.sft = rows.length\n pending.push({ path, write: () => writeFile(path, toJsonl(rows)) })\n } else if (format === 'verifiers') {\n // The declared disposition drives the exporter: 'zero-and-flag' ships\n // gated lines and honest failures as labeled negatives at reward 0.\n const outputs = toVerifiersRolloutOutputs(kept, {\n gatedLines: FORMAT_GATE_DISPOSITION.verifiers,\n })\n gate.byFormat.verifiers = measureFormatGate(gated, releaseRowRefs.verifiers(outputs))\n formatCounts.verifiers = outputs.length\n pending.push({ path, write: () => writeFile(path, toJsonl(outputs)) })\n } else {\n const items = toRftItems(kept, { gatedLines: FORMAT_GATE_DISPOSITION.rft })\n gate.byFormat.rft = measureFormatGate(gated, releaseRowRefs.rft(items))\n formatCounts.rft = items.length\n pending.push({ path, write: () => writeFile(path, toJsonl(items)) })\n }\n }\n\n assertGateReport(gate)\n\n for (const { path, write } of pending) {\n await mkdir(dirname(path), { recursive: true })\n await write()\n files.push(path)\n }\n\n const reportPath = join(options.out, 'scrub-report.json')\n await writeFile(reportPath, `${JSON.stringify(report, null, 2)}\\n`)\n files.push(reportPath)\n\n const cardPath = join(options.out, 'README.md')\n await writeFile(\n cardPath,\n buildDatasetCard({\n lines: kept,\n formats: options.formats,\n includeProposers: options.includeProposers,\n sourceFiles: inputs.map((input) => basename(input)),\n scrubTotals: report.totals,\n excluded: report.excluded,\n formatCounts,\n gate,\n }),\n )\n files.push(cardPath)\n\n return { inputs, read, kept: kept.length, scrub: report, formatCounts, gate, files }\n}\n\nexport function planPushCommand(repo: string, outDir: string): string[] {\n return ['huggingface-cli', 'upload', repo, outDir, '.', '--repo-type', 'dataset']\n}\n\nexport function pushDataset(repo: string, outDir: string): void {\n const found = spawnSync('which', ['huggingface-cli'], { stdio: 'ignore' })\n if (found.status !== 0) {\n throw new Error(\n 'huggingface-cli not found on PATH — install huggingface_hub[cli] before --push',\n )\n }\n if (!process.env.HF_TOKEN) {\n throw new Error('HF_TOKEN not present in env — refusing to push')\n }\n const [command, ...args] = planPushCommand(repo, outDir) as [string, ...string[]]\n // Token stays in the inherited env; it is never echoed or interpolated.\n const run = spawnSync(command, args, { stdio: 'inherit' })\n if (run.status !== 0) throw new Error(`huggingface-cli upload exited ${String(run.status)}`)\n}\n\nexport interface RolloutReleaseCliArgs extends BuildOptions {\n inputs: string[]\n push: string | null\n}\n\nexport const ROLLOUT_RELEASE_USAGE =\n 'usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]'\n\nexport function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs {\n const args: RolloutReleaseCliArgs = {\n inputs: [],\n out: '',\n formats: [...RELEASE_FORMATS],\n includeProposers: false,\n push: null,\n }\n for (let i = 0; i < argv.length; i++) {\n const arg = argv[i]!\n if (arg === '--out') {\n args.out = argv[++i] ?? ''\n } else if (arg === '--formats') {\n const raw = (argv[++i] ?? '').split(',').filter(Boolean)\n for (const format of raw) {\n if (!RELEASE_FORMATS.includes(format as ReleaseFormat)) {\n throw new Error(\n `unknown format \"${format}\" — expected one of ${RELEASE_FORMATS.join(',')}`,\n )\n }\n }\n args.formats = raw as ReleaseFormat[]\n } else if (arg === '--include-proposers') {\n args.includeProposers = true\n } else if (arg === '--push') {\n args.push = argv[++i] ?? null\n } else if (arg.startsWith('--')) {\n throw new Error(`unknown flag \"${arg}\"`)\n } else {\n args.inputs.push(arg)\n }\n }\n if (args.inputs.length === 0 || !args.out) {\n throw new Error(ROLLOUT_RELEASE_USAGE)\n }\n if (args.push !== null && !/^[\\w.-]+\\/[\\w.-]+$/.test(args.push)) {\n throw new Error(`--push expects <org/name>, got \"${args.push}\"`)\n }\n return args\n}\n\n/** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */\nexport async function runRolloutReleaseCli(argv: string[]): Promise<number> {\n let args: RolloutReleaseCliArgs\n try {\n args = parseRolloutReleaseArgs(argv)\n } catch (error) {\n process.stderr.write(`${error instanceof Error ? error.message : String(error)}\\n`)\n return 2\n }\n const summary = await buildHfDataset(args.inputs, args)\n process.stdout.write(\n `${JSON.stringify(\n {\n read: summary.read,\n kept: summary.kept,\n formatCounts: summary.formatCounts,\n gate: summary.gate,\n scrub: summary.scrub,\n },\n null,\n 2,\n )}\\n`,\n )\n process.stdout.write(`dataset → ${args.out} (${summary.files.length} files)\\n`)\n if (args.push !== null) {\n pushDataset(args.push, args.out)\n process.stdout.write(`pushed → ${args.push}\\n`)\n }\n return 0\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;AAoBA,SAAS,UAAU,OAA8B;CAC/C,KAAK,MAAM,CAAC,GAAG,SAAS,MAAM,QAAQ,GAAG,kBAAkB,MAAM,iBAAiB,EAAE,EAAE;CACtF,OAAO,MAAM,KAAK,SAAS,KAAK,UAAU,IAAI,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,MAAM,SAAS,IAAI,OAAO;AAC3F;;AAGA,eAAsB,mBAAmB,MAAc,OAAqC;CAC1F,MAAM,UAAU,UAAU,KAAK;CAC/B,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAC9C,MAAM,UAAU,MAAM,OAAO;AAC/B;;AAGA,eAAsB,mBAAmB,MAAc,OAAqC;CAC1F,IAAI,MAAM,WAAW,GAAG;CACxB,MAAM,UAAU,UAAU,KAAK;CAC/B,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAC9C,MAAM,WAAW,MAAM,OAAO;AAChC;;;;;;;;;;;AAYA,eAAsB,kBAAkB,MAA4C;CAClF,OAAO,UAAU,OAAO,QAAQ,YAAY,aAAa,QAAQ,OAAO,CAAC;AAC3E;;;;;;;;;;;;;;;AAgBA,eAAsB,mBAAmB,MAAsC;CAC7E,OAAO,UAAU,OAAO,QAAQ,YAAyB;EACvD,kBAAkB,QAAQ,OAAO;EACjC,OAAO;CACT,CAAC;AACH;AAEA,eAAe,UACb,MACA,OACc;CACd,MAAM,MAAM,MAAM,SAAS,MAAM,MAAM;CACvC,MAAM,QAAa,CAAC;CACpB,MAAM,WAAW,IAAI,MAAM,IAAI;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,SAAS,QAAQ,KAAK;EACxC,MAAM,OAAO,SAAS;EACtB,IAAI,CAAC,MAAM,KAAK,GAAG;EACnB,IAAI;EACJ,IAAI;GACF,SAAS,KAAK,MAAM,IAAI;EAC1B,SAAS,OAAO;GACd,MAAM,IAAI,MACR,GAAG,KAAK,GAAG,IAAI,EAAE,qBAAqB,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,GAC7F;EACF;EACA,MAAM,KAAK,MAAM,QAAQ,GAAG,KAAK,GAAG,IAAI,GAAG,CAAC;CAC9C;CACA,OAAO;AACT;;;ACtCA,MAAa,0BAAkE;CAC7E,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;;AA2GA,SAAgB,gBAAgB,OAAkD;CAChF,OAAO,IAAI,IAAI,MAAM,QAAQ,SAAS,KAAK,QAAQ,cAAc,CAAC,CAAC,KAAK,SAAS,KAAK,UAAU,CAAC;AACnG;;;;;;;AAQA,MAAa,iBAAiB;CAI5B,MAAM,SACJ,KAAK,KAAK,SAAS;EACjB,YAAY,IAAI,SAAS;EACzB,QAAQ,IAAI,SAAS;EACrB,mBAAmB,IAAI,SAAS;CAClC,EAAE;CAIJ,YAAY,SACV,KAAK,KAAK,SAAS;EACjB,YAAY,IAAI,KAAK;EACrB,QAAQ,IAAI;EACZ,UAAU,IAAI;EACd,mBAAmB,IAAI,KAAK;CAC9B,EAAE;CAGJ,MAAM,SACJ,KAAK,KAAK,SAAS;EACjB,YAAY,IAAI,UAAU;EAC1B,QAAQ,IAAI,UAAU;EACtB,UAAU,IAAI,UAAU;EACxB,mBAAmB,IAAI,UAAU;CACnC,EAAE;CACJ,MAAM,UACJ,MAAM,KAAK,UAAU;EACnB,YAAY,KAAK;EACjB,QAAQ,KAAK,QAAQ;EAIrB,UAAU;GAAE,SAAS,KAAK,QAAQ;GAAS,SAAS,KAAK,QAAQ;EAAQ;EAGzE,cAAc,sBAAsB,KAAK,KAAK;EAC9C,mBAAmB,KAAK,QAAQ,qBAAqB;CACvD,EAAE;AACN;;AAGA,SAAS,kBAAkB,OAAgB,MAAiC;CAC1E,IAAI,OAAO,UAAU,UACnB,OAAO,OAAO,SAAS,KAAK,KAAK,QAAQ,IAAI,CAAC;EAAE;EAAM;CAAM,CAAC,IAAI,CAAC;CAEpE,IAAI,MAAM,QAAQ,KAAK,GACrB,OAAO,MAAM,SAAS,MAAM,MAAM,kBAAkB,MAAM,GAAG,KAAK,GAAG,EAAE,EAAE,CAAC;CAE5E,IAAI,OAAO,UAAU,YAAY,UAAU,MACzC,OAAO,OAAO,QAAQ,KAAK,CAAC,CAAC,SAAS,CAAC,KAAK,WAC1C,kBAAkB,OAAO,SAAS,KAAK,MAAM,GAAG,KAAK,GAAG,KAAK,CAC/D;CAEF,OAAO,CAAC;AACV;;AAGA,SAAgB,kBACd,OACA,MACkB;CAClB,MAAM,UAAU,KAAK,QAAQ,QAAQ,MAAM,IAAI,IAAI,UAAU,CAAC;CAC9D,MAAM,UAAU,QAAQ,KAAK,QAAQ,IAAI,MAAM,CAAC,CAAC,QAAQ,MAAmB,MAAM,IAAI;CACtF,MAAM,WAAW,QAAQ,SAAS,QAAQ,kBAAkB,IAAI,UAAU,EAAE,CAAC;CAC7E,MAAM,eAAe,QAAQ,SAAS,QAAQ,kBAAkB,IAAI,cAAc,OAAO,CAAC;CAC1F,MAAM,aAAa,KAChB,QAAQ,QAAQ,IAAI,sBAAsB,KAAK,CAAC,CAChD,KAAK,QAAQ,IAAI,MAAM,CAAC,CACxB,QAAQ,MAAmB,MAAM,QAAQ,IAAI,CAAC;CACjD,OAAO;EACL,OAAO,MAAM;EACb,SAAS,QAAQ;EACjB,UAAU,MAAM,OAAO,QAAQ;EAC/B,kBAAkB,QAAQ,WAAW,IAAI,OAAO,KAAK,IAAI,GAAG,OAAO;EACnE,oBAAoB,SAAS,QAC1B,MAAM,UAAW,SAAS,QAAQ,MAAM,QAAQ,KAAK,QAAQ,QAAQ,MACtE,IACF;EACA,wBAAwB,WAAW;EACnC,qBAAqB,WAAW,WAAW,IAAI,OAAO,KAAK,IAAI,GAAG,UAAU;EAC5E,wBAAwB,aAAa,QAClC,MAAM,UAAW,SAAS,QAAQ,MAAM,QAAQ,KAAK,QAAQ,QAAQ,MACtE,IACF;CACF;AACF;;;;;;;;;;;;;AAcA,MAAM,kBAEF;CACF,wBAAwB,QAAQ,WAC9B,OAAO,qBAAqB,QAAQ,OAAO,oBAAoB,IAC3D,OACA,mBAAmB,OAAO,KAAK,OAAO,QAAQ,sDAC/B,OAAO,iBAAiB;CAE7C,mBAAmB,QAAQ,WAAW;EACpC,IAAI,OAAO,uBAAuB,MAAM,OAAO;EAC/C,MAAM,EAAE,MAAM,UAAU,OAAO;EAC/B,OACE,mBAAmB,OAAO,KAAK,OAAO,QAAQ,iEACpB,KAAK,KAAK,MAAM;CAK9C;CACA,4BAA4B,QAAQ,WAAW;EAC7C,IAAI,OAAO,2BAA2B,MAAM,OAAO;EACnD,MAAM,EAAE,MAAM,UAAU,OAAO;EAC/B,OACE,mBAAmB,OAAO,KAAK,OAAO,QAAQ,gHAC2B,KAAK,KAAK,MAAM;CAK7F;CACA,sBAAsB,QAAQ,WAC5B,OAAO,wBAAwB,OAC3B,OACA,mBAAmB,OAAO,KAAK,OAAO,uBAAuB,uCAC9C,OAAO,oBAAoB;AAIlD;;;;;;;;;;;;;;;;AAiBA,SAAgB,iBAAiB,QAA0B;CACzD,KAAK,MAAM,CAAC,QAAQ,WAAW,OAAO,QAAQ,OAAO,QAAQ,GAE1D;EACD,KAAK,MAAM,MAAM,gBAAgB;GAC/B,IAAI,cAAc,iBAAiB,GAAG,CAAC,SAAS,WAAW;GAC3D,MAAM,UAAU,gBAAgB,GAAG,CAAC,QAAQ,MAAM;GAClD,IAAI,YAAY,MAAM,MAAM,IAAI,MAAM,OAAO;EAC/C;EAIA,IAAI,wBAAwB,YAAY,aAAa,OAAO,UAAU,GACpE,MAAM,IAAI,MACR,mBAAmB,OAAO,4CAA4C,OAAO,QAAQ,UAEvF;CAEJ;AACF;;;;;;;;;;;;ACxVA,MAAa,kBAAkB;CAAC;CAAO;CAAa;CAAO;AAAK;;AAIhE,MAAa,eAA8C;CACzD,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;AAEA,MAAM,sBAAqD;CACzD,KAAK;CACL,WACE;CACF,KAAK;CACL,KAAK,UAAU,eAAe;AAChC;AAqBA,SAAS,OAAO,QAAwC;CACtD,OAAO,CAAC,GAAG,IAAI,IAAI,OAAO,QAAQ,MAAmB,MAAM,IAAI,CAAC,CAAC,CAAC,CAAC,KAAK;AAC1E;AAEA,SAAS,aAAa,QAA+B;CACnD,IAAI,WAAW,MAAM,OAAO;CAC5B,OAAO,OAAO,UAAU,MAAM,IAAI,OAAO,MAAM,IAAI,OAAO,QAAQ,CAAC;AACrE;AAEA,SAAS,cAAc,QAAkB,MAA0B;CACjE,OAAO;EACL,KAAK,OAAO,KAAK,KAAK,EAAE;EACxB,KAAK,OAAO,UAAU,KAAK,CAAC,CAAC,KAAK,KAAK,EAAE;EACzC,GAAG,KAAK,KAAK,QAAQ,KAAK,IAAI,KAAK,KAAK,EAAE,GAAG;CAC/C,CAAC,CAAC,KAAK,IAAI;AACb;;AAGA,MAAM,kBAAiD;CACrD,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;AAEA,MAAM,oBAAmD;CACvD,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;AAEA,SAAS,YAAY,SAA0B,MAAkB,YAA4B;CAC3F,MAAM,OAAO,QAAQ,KAAK,WAAW;EACnC,MAAM,SAAS,KAAK,SAAS;EAC7B,OAAO;GACL;GACA,kBAAkB;GAClB,OAAO,QAAQ,WAAW,CAAC;GAC3B,OAAO,QAAQ,YAAY,CAAC;GAC5B,QAAQ,qBAAqB,QAAQ,WAAW,KAAA,IAC5C,MACA,aAAa,OAAO,gBAAgB;GACxC,QAAQ,sBAAsB,OAC1B,MACA,GAAG,OAAO,mBAAmB,KAAK,KAAK,OAAO,mBAAmB;GACrE,gBAAgB;EAClB;CACF,CAAC;CACD,MAAM,iBAAiB,QAAQ,MAC5B,WAAW,wBAAwB,YAAY,eAClD;CACA,OAAO;EACL,kIAAkI,KAAK,WAAW,MAAM,WAAW;EACnK;EACA;EACA;EACA;EACA;EACA,cACE;GACE;GACA;GACA;GACA;GACA;GACA;GACA;EACF,GACA,IACF;EACA;EACA;EACA;EACA;EACA,iBACI,qLACA;CACN,CAAC,CACE,KAAK,IAAI,CAAC,CACV,QAAQ;AACb;AAEA,SAAS,eAAe,OAAwC;CAC9D,MAAM,yBAAS,IAAI,IAAsC;CACzD,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,WAAW,OAAO,IAAI,KAAK,IAAI,qBAAK,IAAI,IAAoB;EAClE,MAAM,MAAM,aAAa,KAAK,QAAQ,MAAM;EAC5C,SAAS,IAAI,MAAM,SAAS,IAAI,GAAG,KAAK,KAAK,CAAC;EAC9C,OAAO,IAAI,KAAK,MAAM,QAAQ;CAChC;CACA,MAAM,OAAmB,CAAC;CAC1B,KAAK,MAAM,QAAQ,eAAe;EAChC,MAAM,WAAW,OAAO,IAAI,IAAI;EAChC,IAAI,CAAC,UAAU;EACf,MAAM,OAAO,CAAC,GAAG,SAAS,KAAK,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM;GAC/C,IAAI,MAAM,QAAQ,OAAO;GACzB,IAAI,MAAM,QAAQ,OAAO;GACzB,OAAO,OAAO,CAAC,IAAI,OAAO,CAAC;EAC7B,CAAC;EACD,KAAK,MAAM,OAAO,MAAM,KAAK,KAAK;GAAC;GAAM;GAAK,OAAO,SAAS,IAAI,GAAG,CAAC;EAAC,CAAC;CAC1E;CACA,OAAO;AACT;AAEA,SAAgB,iBAAiB,QAAmC;CAClE,MAAM,EACJ,OACA,SACA,kBACA,aACA,aACA,UACA,cACA,SACE;CAGJ,iBAAiB,IAAI;CAErB,MAAM,SAAS,OAAO,MAAM,KAAK,SAAS,KAAK,MAAM,CAAC;CACtD,MAAM,cAAc,CAAC,GAAG,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,UAAU,CAAC,CAAC,CAAC,CACnE,QAAQ,MAAmB,MAAM,IAAI,CAAC,CACtC,MAAM,GAAG,MAAM,IAAI,CAAC;CACvB,MAAM,SAAS,OAAO,MAAM,KAAK,SAAS,KAAK,OAAO,KAAK,CAAC;CAC5D,MAAM,YAAY,OAAO,MAAM,KAAK,SAAS,KAAK,OAAO,OAAO,CAAC;CACjE,MAAM,WAAW,OAAO,MAAM,KAAK,SAAS,KAAK,WAAW,OAAO,CAAC;CACpE,MAAM,gBAAgB,OAAO,MAAM,KAAK,SAAS,KAAK,QAAQ,aAAa,CAAC;CAC5E,MAAM,WAAW,MAAM,QAAQ,SAAS,KAAK,SAAS,WAAW,CAAC,CAAC,CAAC;CACpE,MAAM,aAAa,MAAM,QAAQ,SAAS,KAAK,QAAQ,cAAc,CAAC,CAAC;CACvE,IAAI,eAAe,KAAK,YACtB,MAAM,IAAI,MACR,oCAAoC,KAAK,WAAW,6DAC1B,WAAW,mDACvC;CAcF,MAAM,cAAc;EAClB;EACA;EACA;EACA;EAfc,QACb,KAAK,WACJ;GACE,oBAAoB;GACpB;GACA;GACA,iBAAiB,aAAa;EAChC,CAAC,CAAC,KAAK,IAAI,CACb,CAAC,CACA,KAAK,IAOA;EACN;CACF,CAAC,CAAC,KAAK,IAAI;CAEX,MAAM,eAAe,cACnB;EAAC;EAAU;EAAQ;EAAQ;CAAU,GACrC,QAAQ,KAAK,WAAW;EACtB;EACA,KAAK,aAAa,QAAQ;EAC1B,OAAO,aAAa,WAAW,CAAC;EAChC,oBAAoB;CACtB,CAAC,CACH;CAEA,MAAM,cAAc,cAAc;EAAC;EAAQ;EAAU;CAAO,GAAG,eAAe,KAAK,CAAC;CAEpF,MAAM,aAAa,cACjB,CAAC,QAAQ,UAAU,GACnB,OAAO,QAAQ,WAAW,CAAC,CAAC,KAAK,CAAC,MAAM,WAAW,CAAC,MAAM,OAAO,KAAK,CAAC,CAAC,CAC1E;CAEA,MAAM,eAAe,mBACjB,uHACA,8CAA8C,SAAS,UAAU;CAErE,OAAO,GAAG,YAAY;;;;gIAIwG,eAAe;;6NAE8E,SAAS,SAAS;;;;EAI7O,aAAa;;;;EAIb,YAAY,SAAS,MAAM,MAAM,MAAM,EAAE;;eAE5B,eAAe;;;;;;;sBAOR,cAAc,KAAK,SAAS,KAAK,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;;;qPAGyK,SAAS;;;;;;;oBAO1O,YAAY,KAAK,SAAS,KAAK,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;aAC3D,OAAO,KAAK,OAAO,KAAK,GAAG,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;iBACvC,YAAY,KAAK,IAAI,EAAE;YAC5B,OAAO,KAAK,MAAM,KAAK,EAAE,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;eACtC,UAAU,KAAK,MAAM,KAAK,EAAE,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;mBACxC,SAAS,KAAK,IAAI,EAAE;2EACoC,cAAc,KAAK,MAAM,KAAK,EAAE,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;;EAEzH,aAAa;;;;;;;;;;;;EAYb,YAAY;;eAEC,MAAM,OAAO;;;;;;EAM1B,WAAW;;;;;;;;;;;;;;;;;AAiBb;;;;;;;;;;;;;;AC/RA,MAAa,cAAoC;CAC/C;EAEE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;CACA;EAGE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;CACA;EAEE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;CACA;EAGE,MAAM;EACN,SACE;EACF,UAAU,QAAQ,WAAW,GAAG,OAAO;CACzC;CACA;EAGE,MAAM;EACN,SACE;EACF,UAAU,QAAQ,SAAS,GAAG,KAAK;CACrC;CACA;EACE,MAAM;EACN,SAAS;EACT,UAAU,QAAQ,WAAW,GAAG,OAAO;CACzC;CACA;EAGE,MAAM;EACN,SACE;EACF,eAAe;CACjB;CACA;EAGE,MAAM;EACN,SAAS;EACT,UAAU,QAAQ,WAAW,GAAG,UAAU,GAAG;CAC/C;CACA;EAGE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;AACF;AAKA,SAAgB,mBAAgC;CAC9C,OAAO,OAAO,YAAY,YAAY,KAAK,SAAS,CAAC,KAAK,MAAM,CAAC,CAAC,CAAC;AACrE;AAEA,SAAgB,eAAe,MAAmB,MAAgC;CAChF,KAAK,MAAM,CAAC,MAAM,UAAU,OAAO,QAAQ,IAAI,GAAG,KAAK,SAAS,KAAK,SAAS,KAAK;CACnF,OAAO;AACT;AAEA,SAAgB,UAAU,MAAc,QAA6B;CACnE,IAAI,MAAM;CACV,KAAK,MAAM,QAAQ,aACjB,MAAM,IAAI,QAAQ,KAAK,UAAU,OAAe,GAAG,SAAoB;EACrE,OAAO,KAAK,SAAS,OAAO,KAAK,SAAS,KAAK;EAC/C,OAAO,KAAK,QAAQ,OAAO,OAAO,KAAK,OAAO,WAAW,KAAK,KAAK,KAAA,CAAS;CAC9E,CAAC;CAEH,OAAO;AACT;AAEA,SAAS,WAAW,OAAgB,QAA8B;CAChE,IAAI,OAAO,UAAU,UAAU,OAAO,UAAU,OAAO,MAAM;CAC7D,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO,MAAM,KAAK,SAAS,WAAW,MAAM,MAAM,CAAC;CAC7E,IAAI,UAAU,QAAQ,OAAO,UAAU,UAAU;EAC/C,MAAM,MAA+B,CAAC;EACtC,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,KAAK,GAAG,IAAI,OAAO,WAAW,MAAM,MAAM;EACnF,OAAO;CACT;CACA,OAAO;AACT;;;;;;;;;AAUA,SAAgB,iBAAiB,MAAyB,QAAwC;CAChG,OAAO,aAAa,WAAW,MAAM,MAAM,GAAG,yBAAyB,KAAK,YAAY;AAC1F;AAEA,SAAgB,WAAW,OAGzB;CACA,MAAM,SAAS,iBAAiB;CAChC,OAAO;EAAE,OAAO,MAAM,KAAK,SAAS,iBAAiB,MAAM,MAAM,CAAC;EAAG;CAAO;AAC9E;;;;;;;;AASA,SAAgB,uBAAuB,MAAsB;CAC3D,OAAO,UAAU,MAAM,iBAAiB,CAAC;AAC3C;;;;;;;;;;;;;;;;;;;;AC1FA,eAAsB,eACpB,QACA,SACuB;CACvB,IAAI,OAAO,WAAW,GAAG,MAAM,IAAI,MAAM,wBAAwB;CACjE,IAAI,QAAQ,QAAQ,WAAW,GAAG,MAAM,IAAI,MAAM,qBAAqB;CAEvE,MAAM,SAAsB;EAC1B,OAAO,CAAC;EACR,QAAQ,iBAAiB;EACzB,UAAU;GAAE,WAAW;GAAG,UAAU;EAAE;CACxC;CACA,MAAM,OAA4B,CAAC;CACnC,IAAI,OAAO;CAEX,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,QAAQ,MAAM,kBAAkB,KAAK;EAC3C,QAAQ,MAAM;EAYd,MAAM,WAAW,WAXC,MAAM,QAAQ,SAAS;GACvC,IAAI,CAAC,iBAAiB,KAAK,KAAK,KAAK,GAAG;IACtC,OAAO,SAAS,YAAY;IAC5B,OAAO;GACT;GACA,IAAI,CAAC,QAAQ,oBAAoB,KAAK,SAAS,YAAY;IACzD,OAAO,SAAS,aAAa;IAC7B,OAAO;GACT;GACA,OAAO;EACT,CACoC,CAAC;EACrC,OAAO,MAAM,SAAS,SAAS;EAC/B,eAAe,OAAO,QAAQ,SAAS,MAAM;EAC7C,KAAK,KAAK,GAAG,SAAS,KAAK;CAC7B;CAEA,MAAM,eAAuD,CAAC;CAC9D,MAAM,QAAkB,CAAC;CACzB,MAAM,QAAQ,gBAAgB,IAAI;CAClC,MAAM,OAAmB;EAAE,YAAY,MAAM;EAAM,UAAU,CAAC;CAAE;CAMhE,MAAM,UAA+D,CAAC;CAEtE,KAAK,MAAM,UAAU,QAAQ,SAAS;EACpC,MAAM,OAAO,KAAK,QAAQ,KAAK,aAAa,OAAO;EACnD,IAAI,WAAW,OAAO;GAEpB,KAAK,SAAS,MAAM,kBAAkB,OAAO,eAAe,IAAI,IAAI,CAAC;GACrE,aAAa,MAAM,KAAK;GACxB,QAAQ,KAAK;IAAE;IAAM,aAAa,mBAAmB,MAAM,IAAI;GAAE,CAAC;EACpE,OAAO,IAAI,WAAW,OAAO;GAC3B,MAAM,OAAO,UAAU,IAAI;GAC3B,KAAK,SAAS,MAAM,kBAAkB,OAAO,eAAe,IAAI,IAAI,CAAC;GACrE,aAAa,MAAM,KAAK;GACxB,QAAQ,KAAK;IAAE;IAAM,aAAa,UAAU,MAAM,QAAQ,IAAI,CAAC;GAAE,CAAC;EACpE,OAAO,IAAI,WAAW,aAAa;GAGjC,MAAM,UAAU,0BAA0B,MAAM,EAC9C,YAAY,wBAAwB,UACtC,CAAC;GACD,KAAK,SAAS,YAAY,kBAAkB,OAAO,eAAe,UAAU,OAAO,CAAC;GACpF,aAAa,YAAY,QAAQ;GACjC,QAAQ,KAAK;IAAE;IAAM,aAAa,UAAU,MAAM,QAAQ,OAAO,CAAC;GAAE,CAAC;EACvE,OAAO;GACL,MAAM,QAAQ,WAAW,MAAM,EAAE,YAAY,wBAAwB,IAAI,CAAC;GAC1E,KAAK,SAAS,MAAM,kBAAkB,OAAO,eAAe,IAAI,KAAK,CAAC;GACtE,aAAa,MAAM,MAAM;GACzB,QAAQ,KAAK;IAAE;IAAM,aAAa,UAAU,MAAM,QAAQ,KAAK,CAAC;GAAE,CAAC;EACrE;CACF;CAEA,iBAAiB,IAAI;CAErB,KAAK,MAAM,EAAE,MAAM,WAAW,SAAS;EACrC,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;EAC9C,MAAM,MAAM;EACZ,MAAM,KAAK,IAAI;CACjB;CAEA,MAAM,aAAa,KAAK,QAAQ,KAAK,mBAAmB;CACxD,MAAM,UAAU,YAAY,GAAG,KAAK,UAAU,QAAQ,MAAM,CAAC,EAAE,GAAG;CAClE,MAAM,KAAK,UAAU;CAErB,MAAM,WAAW,KAAK,QAAQ,KAAK,WAAW;CAC9C,MAAM,UACJ,UACA,iBAAiB;EACf,OAAO;EACP,SAAS,QAAQ;EACjB,kBAAkB,QAAQ;EAC1B,aAAa,OAAO,KAAK,UAAU,SAAS,KAAK,CAAC;EAClD,aAAa,OAAO;EACpB,UAAU,OAAO;EACjB;EACA;CACF,CAAC,CACH;CACA,MAAM,KAAK,QAAQ;CAEnB,OAAO;EAAE;EAAQ;EAAM,MAAM,KAAK;EAAQ,OAAO;EAAQ;EAAc;EAAM;CAAM;AACrF;AAEA,SAAgB,gBAAgB,MAAc,QAA0B;CACtE,OAAO;EAAC;EAAmB;EAAU;EAAM;EAAQ;EAAK;EAAe;CAAS;AAClF;AAEA,SAAgB,YAAY,MAAc,QAAsB;CAE9D,IADc,UAAU,SAAS,CAAC,iBAAiB,GAAG,EAAE,OAAO,SAAS,CAChE,CAAC,CAAC,WAAW,GACnB,MAAM,IAAI,MACR,gFACF;CAEF,IAAI,CAAC,QAAQ,IAAI,UACf,MAAM,IAAI,MAAM,gDAAgD;CAElE,MAAM,CAAC,SAAS,GAAG,QAAQ,gBAAgB,MAAM,MAAM;CAEvD,MAAM,MAAM,UAAU,SAAS,MAAM,EAAE,OAAO,UAAU,CAAC;CACzD,IAAI,IAAI,WAAW,GAAG,MAAM,IAAI,MAAM,iCAAiC,OAAO,IAAI,MAAM,GAAG;AAC7F;AAOA,MAAa,wBACX;AAEF,SAAgB,wBAAwB,MAAuC;CAC7E,MAAM,OAA8B;EAClC,QAAQ,CAAC;EACT,KAAK;EACL,SAAS,CAAC,GAAG,eAAe;EAC5B,kBAAkB;EAClB,MAAM;CACR;CACA,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,MAAM,KAAK;EACjB,IAAI,QAAQ,SACV,KAAK,MAAM,KAAK,EAAE,MAAM;OACnB,IAAI,QAAQ,aAAa;GAC9B,MAAM,OAAO,KAAK,EAAE,MAAM,GAAA,CAAI,MAAM,GAAG,CAAC,CAAC,OAAO,OAAO;GACvD,KAAK,MAAM,UAAU,KACnB,IAAI,CAAC,gBAAgB,SAAS,MAAuB,GACnD,MAAM,IAAI,MACR,mBAAmB,OAAO,sBAAsB,gBAAgB,KAAK,GAAG,GAC1E;GAGJ,KAAK,UAAU;EACjB,OAAO,IAAI,QAAQ,uBACjB,KAAK,mBAAmB;OACnB,IAAI,QAAQ,UACjB,KAAK,OAAO,KAAK,EAAE,MAAM;OACpB,IAAI,IAAI,WAAW,IAAI,GAC5B,MAAM,IAAI,MAAM,iBAAiB,IAAI,EAAE;OAEvC,KAAK,OAAO,KAAK,GAAG;CAExB;CACA,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC,KAAK,KACpC,MAAM,IAAI,MAAM,qBAAqB;CAEvC,IAAI,KAAK,SAAS,QAAQ,CAAC,qBAAqB,KAAK,KAAK,IAAI,GAC5D,MAAM,IAAI,MAAM,mCAAmC,KAAK,KAAK,EAAE;CAEjE,OAAO;AACT;;AAGA,eAAsB,qBAAqB,MAAiC;CAC1E,IAAI;CACJ,IAAI;EACF,OAAO,wBAAwB,IAAI;CACrC,SAAS,OAAO;EACd,QAAQ,OAAO,MAAM,GAAG,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,EAAE,GAAG;EAClF,OAAO;CACT;CACA,MAAM,UAAU,MAAM,eAAe,KAAK,QAAQ,IAAI;CACtD,QAAQ,OAAO,MACb,GAAG,KAAK,UACN;EACE,MAAM,QAAQ;EACd,MAAM,QAAQ;EACd,cAAc,QAAQ;EACtB,MAAM,QAAQ;EACd,OAAO,QAAQ;CACjB,GACA,MACA,CACF,EAAE,GACJ;CACA,QAAQ,OAAO,MAAM,aAAa,KAAK,IAAI,IAAI,QAAQ,MAAM,OAAO,UAAU;CAC9E,IAAI,KAAK,SAAS,MAAM;EACtB,YAAY,KAAK,MAAM,KAAK,GAAG;EAC/B,QAAQ,OAAO,MAAM,YAAY,KAAK,KAAK,GAAG;CAChD;CACA,OAAO;AACT"}
|
|
1
|
+
{"version":3,"file":"hf-dataset-D8_RNIis.js","names":[],"sources":["../src/rollout/ledger.ts","../src/rollout/release/gate-report.ts","../src/rollout/release/card.ts","../src/rollout/release/scrub.ts","../src/rollout/release/hf-dataset.ts"],"sourcesContent":["/**\n * Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`\n * lines. Writes validate BEFORE touching disk (a bad line never lands);\n * reads validate line-by-line and fail loud with the line number, because a\n * silently-skipped rollout is a corrupted dataset.\n *\n * \"Validate\" includes the anti-Goodhart invariant (a realness-gated line may\n * not carry a positive reward), so a poisoned line can neither enter a ledger\n * nor leave one.\n *\n * Two read modes, matching the two write-side row classes: `readRolloutLedger`\n * re-validates under the mint policy (training data), `readRolloutJournal`\n * under the write policy (supervision journals, whose unscreened positive\n * rewards are writable and must stay readable).\n */\n\nimport { appendFile, mkdir, readFile, writeFile } from 'node:fs/promises'\nimport { dirname } from 'node:path'\nimport { assertMinted, assertRolloutLine, type MintedRolloutLine, type RolloutLine } from './schema'\n\nfunction serialize(lines: RolloutLine[]): string {\n for (const [i, line] of lines.entries()) assertRolloutLine(line, `rollout line [${i}]`)\n return lines.map((line) => JSON.stringify(line)).join('\\n') + (lines.length > 0 ? '\\n' : '')\n}\n\n/** Replace the ledger file with exactly `lines`. */\nexport async function writeRolloutLedger(path: string, lines: RolloutLine[]): Promise<void> {\n const payload = serialize(lines)\n await mkdir(dirname(path), { recursive: true })\n await writeFile(path, payload)\n}\n\n/** Append `lines` to the ledger file (created if absent). */\nexport async function appendRolloutLines(path: string, lines: RolloutLine[]): Promise<void> {\n if (lines.length === 0) return\n const payload = serialize(lines)\n await mkdir(dirname(path), { recursive: true })\n await appendFile(path, payload)\n}\n\n/**\n * Read and validate every line. Throws on the first malformed/invalid line\n * (with its 1-based line number) — fail-closed, never a silent drop.\n *\n * Validation includes the anti-Goodhart invariant, which is why the result is\n * `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this\n * process from outside the type system (another run, another machine, a\n * hand-edited JSONL), so this read is the runtime boundary where a poisoned\n * line is refused rather than exported.\n */\nexport async function readRolloutLedger(path: string): Promise<MintedRolloutLine[]> {\n return readLines(path, (parsed, context) => assertMinted(parsed, context))\n}\n\n/**\n * Read a ledger under the WRITE-side policy (`validateRolloutLine`), which\n * omits the unscreened-reward check. `writeRolloutLedger` accepts a\n * supervision-journal row (`realness_screened: false` with a positive reward\n * — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says\n * such rows \"must stay writable, readable and reportable\"; a read API that\n * only re-validated under `assertMinted` made every such file unreadable —\n * write-accepted but read-refused is a data-loss trap.\n *\n * The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here\n * can reach a training exporter without passing `assertMinted`, so the\n * promotion gate (which DOES enforce unscreened-reward) is exactly as closed\n * as before. Use `readRolloutLedger` when the file is training data.\n */\nexport async function readRolloutJournal(path: string): Promise<RolloutLine[]> {\n return readLines(path, (parsed, context): RolloutLine => {\n assertRolloutLine(parsed, context)\n return parsed\n })\n}\n\nasync function readLines<T>(\n path: string,\n admit: (parsed: unknown, context: string) => T,\n): Promise<T[]> {\n const raw = await readFile(path, 'utf8')\n const lines: T[] = []\n const rawLines = raw.split('\\n')\n for (let i = 0; i < rawLines.length; i++) {\n const text = rawLines[i]\n if (!text?.trim()) continue\n let parsed: unknown\n try {\n parsed = JSON.parse(text)\n } catch (error) {\n throw new Error(\n `${path}:${i + 1}: malformed JSON — ${error instanceof Error ? error.message : String(error)}`,\n )\n }\n lines.push(admit(parsed, `${path}:${i + 1}`))\n }\n return lines\n}\n","/**\n * Per-format accounting of the anti-Goodhart gate for one dataset release.\n *\n * The defect this exists to make impossible: a dataset card that STATES what\n * the gate does while the build does something else. A sentence in a README is\n * a claim about bytes it never reads, so it drifts the moment an exporter\n * changes — and the drift ships to whoever downloads the dataset.\n *\n * So the card is not allowed to assert anything about the gate. The build\n * measures the rows it is ABOUT TO WRITE (`measureFormatGate`), the measurement\n * is checked against the declared per-format disposition (`assertGateReport`,\n * which throws rather than warns), and the card renders only numbers handed to\n * it. A card that disagrees with its own data files cannot be produced without\n * failing the build first.\n *\n * The dispositions themselves are the release policy, stated once as data:\n *\n * sft EXCLUDE — an SFT row is an imitation target. A gamed\n * trajectory must never be imitated, at any weight.\n * verifiers ZERO_AND_FLAG — reward is a signed learning signal here, so a\n * gamed trajectory at reward 0 is a correct\n * negative. Dropping it would bias the negative\n * population toward honest failures and leave a\n * trainer no example of what gaming looks like\n * when it is penalized.\n * rft ZERO_AND_FLAG — RFT re-samples the completion; only the prompt\n * and the grader's `reference.*` verdict ship, so\n * nothing gamed is imitated. The flag is what lets\n * a grader author skip the instance.\n * raw ZERO_AND_FLAG — a faithful audit dump. Removing rows from it\n * would defeat its only purpose, and the gated\n * row is the one an auditor most wants.\n *\n * `ZERO_AND_FLAG` is never `reward: 0` alone. Zeroing without the label makes a\n * faked success indistinguishable from an honest failure — it hides the gamed\n * population from the buyer instead of disclosing it. Every included format\n * carries `realness_gated` on the row itself.\n *\n * And `ZERO_AND_FLAG` means the whole outcome, not the scalar. A gated row that\n * ships `reward: 0` beside the per-layer verifier scores the reward was\n * computed from has not been zeroed in any sense a trainer respects; the\n * accounting therefore measures every reward-derived number each format writes,\n * not just the one field.\n */\n\nimport type { RftItem, SftRow, VerifiersRolloutOutput } from '../exporters'\nimport {\n GATE_CHECK_IDS,\n GATE_POLICIES,\n type GateCheckId,\n undeclaredStepPayload,\n} from '../gate-checks'\nimport type { MintedRolloutLine } from '../schema'\nimport type { ReleaseFormat } from './card'\n\n/** What a format does with a line the realness gate flagged. */\nexport type GateDisposition = 'exclude' | 'zero-and-flag'\n\nexport const FORMAT_GATE_DISPOSITION: Record<ReleaseFormat, GateDisposition> = {\n sft: 'exclude',\n verifiers: 'zero-and-flag',\n rft: 'zero-and-flag',\n raw: 'zero-and-flag',\n}\n\n/** What the gate accounting reads off an emitted row, per format. */\nexport interface ReleaseRowRef {\n rollout_id: string\n reward: number | null\n /**\n * The rest of the row that was DERIVED from the reward — the per-layer score\n * dict, the judge verdict record, whatever this format ships beside the\n * scalar. Walked for positive numbers, so the certification is about the\n * whole outcome rather than one field.\n *\n * Absent when the format's row carries nothing but the scalar. NOT the whole\n * row: `cost.tokens_in`, `wall_s` and `total_steps` are positive numbers that\n * have nothing to do with the reward, and a certification that flags them is\n * a certification nobody can act on.\n */\n evidence?: unknown\n /**\n * The screen claim AS EMITTED — read off the row, not off the line it came\n * from, because what ships is what matters. Required, not optional: an\n * optional field is how a format quietly opts out of the check that reads it,\n * and every emitted row shape carries `RealnessLabels` precisely so no adapter\n * has to.\n */\n realness_screened: boolean | null\n /**\n * The part of an emitted `steps[]` the wire format does not declare.\n *\n * Separate from `evidence` because the declared step fields are FULL of\n * legitimate positive numbers — `durationMs`, `llm_call_count`,\n * `prompt_token_ids` — and a certification that flags those is one nobody can\n * act on. Only the undeclared remainder is unclassified reward-bearing\n * payload, which is the same partition the check applies.\n *\n * Set only by formats whose row carries steps: today `raw` alone.\n */\n stepEvidence?: unknown\n}\n\n/** A positive number found inside an emitted gated row, with where it was. */\nexport interface EmittedEvidence {\n /** JSON-ish path from the row's evidence root, e.g. `metrics['layer.tests']`. */\n path: string\n value: number\n}\n\nexport interface FormatGateCounts {\n /** Gated lines that reached this format's exporter. */\n input: number\n /** Gated rows the format actually wrote. */\n emitted: number\n /**\n * Gated lines this format did not write. Not all of these are the gate:\n * `verifiers` also drops gap lines (empty transcript) and `rft` drops lines\n * with no prompt turn, so an excluded count can mix both causes.\n */\n excluded: number\n /** Highest reward on an emitted gated row; `null` when none was emitted. */\n maxEmittedReward: number | null\n /**\n * The largest positive number found in the reward-DERIVED payload of an\n * emitted gated row, and its path; `null` when there is none.\n *\n * This column exists because the release once certified CLEAN while leaking.\n * `assertGateReport` inspected `outcome.reward` alone, so a gated row shipping\n * `reward: 0` next to `metrics['layer.tests']: 1` — the deterministic verifier\n * score the reward was computed from, and the per-rubric score dict of the\n * Prime Intellect verifiers format — passed, and the card rendered \"max reward\n * | 0\" over a file that carried the gamed signal at full value. A wrong\n * certification is worse than the leak: it is the leak plus a document saying\n * there isn't one.\n */\n maxEmittedEvidence: EmittedEvidence | null\n /**\n * Rows this format wrote carrying a positive reward whose producer DECLARED\n * that no authenticity screen ever ran on it (`realness_screened: false`).\n *\n * Measured over EVERY emitted row, not just the gated ones: an unscreened\n * reward is by definition one the gate never had a verdict on, so it is not in\n * the gated set and a measurement scoped to that set would report 0 forever.\n * `assertMinted` already refuses these, which is exactly why the release still\n * measures them — the last door before a public dataset does not get to assume\n * the earlier doors held.\n */\n unscreenedPositiveRows: number\n /** Highest reward on such a row; `null` when there is none. */\n maxUnscreenedReward: number | null\n /**\n * The largest positive number found in an emitted gated row's UNDECLARED\n * per-step payload, and its path; `null` when there is none.\n *\n * The column exists because the gate read `outcome` and nothing else for\n * three rounds, so a gated line shipping `steps: [{kind, name, reward: 0.86}]`\n * certified clean — the release accounting agreed with the exporter that a\n * per-step reward was not a reward.\n */\n maxEmittedStepEvidence: EmittedEvidence | null\n}\n\nexport interface GateReport {\n /** Gated lines in the release input, after the split/proposer filters. */\n gatedLines: number\n byFormat: Partial<Record<ReleaseFormat, FormatGateCounts>>\n}\n\n/** Rollout ids of every gated line, the key the emitted rows are matched on. */\nexport function gatedRolloutIds(lines: readonly MintedRolloutLine[]): Set<string> {\n return new Set(lines.filter((line) => line.outcome.realness_gated).map((line) => line.rollout_id))\n}\n\n/**\n * Row refs per format. Written as one adapter per format so that the knowledge\n * of WHERE the id and reward live in each published shape sits next to the\n * assertion that uses it — an exporter that moves either field breaks here\n * rather than silently reporting zero gated rows.\n */\nexport const releaseRowRefs = {\n // An SFT row is `{messages, metadata}` and `metadata` holds ids plus the\n // scalar — no reward-derived payload beyond `reward` itself, and the gated\n // disposition is EXCLUDE anyway.\n sft: (rows: readonly SftRow[]): ReleaseRowRef[] =>\n rows.map((row) => ({\n rollout_id: row.metadata.rollout_id,\n reward: row.metadata.reward,\n realness_screened: row.metadata.realness_screened,\n })),\n // `metrics` IS the per-rubric score dict in the Prime Intellect verifiers\n // format — the same numbers the reward was computed from. This is the leak\n // that shipped.\n verifiers: (rows: readonly VerifiersRolloutOutput[]): ReleaseRowRef[] =>\n rows.map((row) => ({\n rollout_id: row.info.rollout_id,\n reward: row.reward,\n evidence: row.metrics,\n realness_screened: row.info.realness_screened,\n })),\n // RFT re-samples the completion, but the grader reads `item.reference.*`, so\n // the verbatim judge verdict is a reward-bearing field on the row.\n rft: (rows: readonly RftItem[]): ReleaseRowRef[] =>\n rows.map((row) => ({\n rollout_id: row.reference.rollout_id,\n reward: row.reference.reward,\n evidence: row.reference.verdict,\n realness_screened: row.reference.realness_screened,\n })),\n raw: (lines: readonly MintedRolloutLine[]): ReleaseRowRef[] =>\n lines.map((line) => ({\n rollout_id: line.rollout_id,\n reward: line.outcome.reward,\n // `provenance.gated_evidence` is deliberately NOT walked: relocating the\n // diagnostics there is what the gate DOES, and the raw config is the\n // audit dump those diagnostics exist for.\n evidence: { metrics: line.outcome.metrics, verdict: line.outcome.verdict },\n // `raw` is the only config that writes the whole line, so it is the only\n // one whose rows can carry a per-step reward.\n stepEvidence: undeclaredStepPayload(line.steps),\n realness_screened: line.outcome.realness_screened ?? null,\n })),\n}\n\n/** Every positive finite number inside a row's reward-derived payload, with its path. */\nfunction positiveNumbersIn(value: unknown, path: string): EmittedEvidence[] {\n if (typeof value === 'number') {\n return Number.isFinite(value) && value > 0 ? [{ path, value }] : []\n }\n if (Array.isArray(value)) {\n return value.flatMap((item, i) => positiveNumbersIn(item, `${path}[${i}]`))\n }\n if (typeof value === 'object' && value !== null) {\n return Object.entries(value).flatMap(([key, child]) =>\n positiveNumbersIn(child, path === '' ? key : `${path}.${key}`),\n )\n }\n return []\n}\n\n/** Measure one format's gated rows from the refs of the rows about to be written. */\nexport function measureFormatGate(\n gated: ReadonlySet<string>,\n refs: readonly ReleaseRowRef[],\n): FormatGateCounts {\n const emitted = refs.filter((ref) => gated.has(ref.rollout_id))\n const rewards = emitted.map((ref) => ref.reward).filter((r): r is number => r !== null)\n const evidence = emitted.flatMap((ref) => positiveNumbersIn(ref.evidence, ''))\n const stepEvidence = emitted.flatMap((ref) => positiveNumbersIn(ref.stepEvidence, 'steps'))\n const unscreened = refs\n .filter((ref) => ref.realness_screened === false)\n .map((ref) => ref.reward)\n .filter((r): r is number => r !== null && r > 0)\n return {\n input: gated.size,\n emitted: emitted.length,\n excluded: gated.size - emitted.length,\n maxEmittedReward: rewards.length === 0 ? null : Math.max(...rewards),\n maxEmittedEvidence: evidence.reduce<EmittedEvidence | null>(\n (best, found) => (best === null || found.value > best.value ? found : best),\n null,\n ),\n unscreenedPositiveRows: unscreened.length,\n maxUnscreenedReward: unscreened.length === 0 ? null : Math.max(...unscreened),\n maxEmittedStepEvidence: stepEvidence.reduce<EmittedEvidence | null>(\n (best, found) => (best === null || found.value > best.value ? found : best),\n null,\n ),\n }\n}\n\n/**\n * The measured form of each canonical gate check, over the rows a release is\n * ABOUT TO WRITE. Returns the failure message, or `null` when the format is\n * clean on that check.\n *\n * TOTAL over `GateCheckId` — this map and `GATE_POLICIES.assertGateReport` are\n * the two things a new check breaks here, so the release certifier cannot be\n * left behind by a check added anywhere else in the package. That is the whole\n * point: for four rounds each guard composed its own subset by hand, and a\n * release certifying CLEAN while leaking is the most expensive version of that\n * mistake, because it is the leak plus a document saying there isn't one.\n */\nconst REPORT_MEASURES: {\n readonly [K in GateCheckId]: (format: ReleaseFormat, counts: FormatGateCounts) => string | null\n} = {\n 'reward-relationship': (format, counts) =>\n counts.maxEmittedReward === null || counts.maxEmittedReward <= 0\n ? null\n : `release format \"${format}\": ${counts.emitted} realness-gated row(s) carry a positive ` +\n `reward (max ${counts.maxEmittedReward}). A run flagged as gamed may not ship a ` +\n 'positive reward in any config.',\n 'gated-evidence': (format, counts) => {\n if (counts.maxEmittedEvidence === null) return null\n const { path, value } = counts.maxEmittedEvidence\n return (\n `release format \"${format}\": ${counts.emitted} realness-gated row(s) carry a positive ` +\n `reward-derived number (${path} = ${value}). Zeroing the scalar is not enough — the ` +\n 'per-layer scores and judge verdict a fabricated reward was computed FROM are the ' +\n 'same signal in component form, and in this format they are read as training input. ' +\n 'They belong in `provenance.gated_evidence` (see `gateGamedOutcome`), not on the row.'\n )\n },\n 'undeclared-step-payload': (format, counts) => {\n if (counts.maxEmittedStepEvidence === null) return null\n const { path, value } = counts.maxEmittedStepEvidence\n return (\n `release format \"${format}\": ${counts.emitted} realness-gated row(s) carry a positive ` +\n `per-step number under a field \\`tangle.rollout.v1\\` does not declare (${path} = ${value}). ` +\n 'A per-step reward is the same training signal as the scalar, in credit-assignment form, ' +\n 'and the exporters copy `steps` through verbatim — so the row ships it beside a `reward` ' +\n 'of 0. It belongs in `provenance.gated_evidence.steps` (see `gateGamedOutcome`).'\n )\n },\n 'unscreened-reward': (format, counts) =>\n counts.maxUnscreenedReward === null\n ? null\n : `release format \"${format}\": ${counts.unscreenedPositiveRows} row(s) carry a positive ` +\n `reward (max ${counts.maxUnscreenedReward}) whose producer declares that NO authenticity ` +\n 'screen ran on it (`realness_screened: false`). Nothing has established those successes ' +\n 'are real, and a published dataset may not present an unqualified verdict as a measured ' +\n 'one. Screen the runs, or publish them at `reward: null`.',\n}\n\n/**\n * Fail the build when the measurement disagrees with the declared policy.\n *\n * Throws, never filters: an emitted positive reward on a gated row means an\n * exporter upstream stopped applying the gate, and silently dropping the row\n * would hide the producer that made it — the producer is the actual defect.\n *\n * Certifies the whole emitted outcome, not `reward` alone. The earlier version\n * checked one field and therefore certified a release CLEAN while its\n * `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in\n * the top-level `metrics` dict — the card then rendered \"max reward | 0\" over\n * exactly that file. A certification that is wrong is worse than an\n * uncertified leak, so the checks it runs are no longer written down here at\n * all: it iterates `GATE_CHECK_IDS` under its own declared policy.\n */\nexport function assertGateReport(report: GateReport): void {\n for (const [format, counts] of Object.entries(report.byFormat) as Array<\n [ReleaseFormat, FormatGateCounts]\n >) {\n for (const id of GATE_CHECK_IDS) {\n if (GATE_POLICIES.assertGateReport[id].kind !== 'enforce') continue\n const failure = REPORT_MEASURES[id](format, counts)\n if (failure !== null) throw new Error(failure)\n }\n // Not a gate check: this one is the RELEASE POLICY (`FORMAT_GATE_DISPOSITION`)\n // rather than the anti-Goodhart invariant — it asks whether the format did\n // what it said it would with gated rows, not whether a row is poisoned.\n if (FORMAT_GATE_DISPOSITION[format] === 'exclude' && counts.emitted > 0) {\n throw new Error(\n `release format \"${format}\" declares gated lines EXCLUDED but wrote ${counts.emitted} ` +\n 'of them.',\n )\n }\n }\n}\n","/**\n * HuggingFace dataset-card (README.md) generation for a rollout-ledger release.\n *\n * The card is a pure function of the SCRUBBED lines plus the release options —\n * no timestamps, no environment reads — so rebuilding from the same ledger\n * yields byte-identical output. It documents the schema, provenance (run ids,\n * generations, the official judge), per-role reward semantics including the\n * inherited/contribution caveat, and a role × reward counts table.\n */\n\nimport { type MintedRolloutLine, ROLLOUT_ROLES, ROLLOUT_SCHEMA, type RolloutRole } from '../schema'\nimport { assertGateReport, FORMAT_GATE_DISPOSITION, type GateReport } from './gate-report'\nimport type { ScrubCounts } from './scrub'\n\nexport const RELEASE_FORMATS = ['sft', 'verifiers', 'rft', 'raw'] as const\nexport type ReleaseFormat = (typeof RELEASE_FORMATS)[number]\n\n/** Format → data file path inside the dataset dir (train split only). */\nexport const FORMAT_FILES: Record<ReleaseFormat, string> = {\n sft: 'sft/train.jsonl',\n verifiers: 'verifiers/train.jsonl',\n rft: 'rft/train.jsonl',\n raw: 'raw/train.jsonl',\n}\n\nconst FORMAT_DESCRIPTIONS: Record<ReleaseFormat, string> = {\n sft: 'Successful trainable-split transcripts (`reward >= 1`, never realness-gated) as `{messages, metadata}` chat JSONL.',\n verifiers:\n 'Prime Intellect verifiers `RolloutOutput`: prompt/completion split at the first assistant turn, plus reward, metrics, tool defs, and token usage.',\n rft: 'OpenAI RFT items: prompt turns plus `reference.*` verdict fields for a grader (completions are re-sampled during RFT).',\n raw: `Full \\`${ROLLOUT_SCHEMA}\\` ledger lines (scrubbed), one per agent invocation.`,\n}\n\nexport interface DatasetCardInputs {\n /** Scrubbed, release-filtered lines (what actually ships). */\n lines: MintedRolloutLine[]\n formats: ReleaseFormat[]\n includeProposers: boolean\n /** Source ledger basenames, for provenance. */\n sourceFiles: string[]\n scrubTotals: ScrubCounts\n excluded: { proposers: number; nonTrain: number }\n formatCounts: Partial<Record<ReleaseFormat, number>>\n /**\n * Per-format anti-Goodhart accounting MEASURED on the rows the build wrote.\n * Required, not optional: the card's only statement about the gate is a\n * render of these numbers, so a card cannot be produced without them and\n * cannot drift from the data files it ships beside.\n */\n gate: GateReport\n}\n\nfunction unique(values: Array<string | null>): string[] {\n return [...new Set(values.filter((v): v is string => v !== null))].sort()\n}\n\nfunction formatReward(reward: number | null): string {\n if (reward === null) return 'null'\n return Number.isInteger(reward) ? String(reward) : reward.toFixed(4)\n}\n\nfunction markdownTable(header: string[], rows: string[][]): string {\n return [\n `| ${header.join(' | ')} |`,\n `| ${header.map(() => '---').join(' | ')} |`,\n ...rows.map((row) => `| ${row.join(' | ')} |`),\n ].join('\\n')\n}\n\n/** Where the anti-Goodhart flag lives on each format's emitted row. */\nconst GATE_FLAG_FIELD: Record<ReleaseFormat, string> = {\n sft: '— (no gated row is written)',\n verifiers: '`info.realness_gated`',\n rft: '`reference.realness_gated`',\n raw: '`outcome.realness_gated`',\n}\n\nconst GATE_POLICY_LABEL: Record<ReleaseFormat, string> = {\n sft: 'EXCLUDED — an SFT row is an imitation target',\n verifiers: 'INCLUDED, reward forced to 0, flagged',\n rft: 'INCLUDED, reward forced to 0, flagged',\n raw: 'INCLUDED verbatim, flagged',\n}\n\nfunction gateSection(formats: ReleaseFormat[], gate: GateReport, totalLines: number): string {\n const rows = formats.map((format) => {\n const counts = gate.byFormat[format]\n return [\n format,\n GATE_POLICY_LABEL[format],\n String(counts?.emitted ?? 0),\n String(counts?.excluded ?? 0),\n counts?.maxEmittedReward === null || counts === undefined\n ? '—'\n : formatReward(counts.maxEmittedReward),\n counts?.maxEmittedEvidence == null\n ? '—'\n : `${counts.maxEmittedEvidence.path} = ${counts.maxEmittedEvidence.value}`,\n GATE_FLAG_FIELD[format],\n ]\n })\n const mixedExclusion = formats.some(\n (format) => FORMAT_GATE_DISPOSITION[format] === 'zero-and-flag',\n )\n return [\n `\\`outcome.realness_gated: true\\` marks a run whose success signal was faked (it satisfied the proxy without doing the work). **${gate.gatedLines} of ${totalLines} lines in this release are gated.**`,\n '',\n 'The gate is enforced, not asserted. A line pairing `realness_gated: true` with a reward above 0 is rejected by the schema validator, so it cannot be read out of a source ledger or written into `raw/train.jsonl` at all; on a gated line the numbers the reward was computed from (`outcome.metrics`, the verbatim judge `outcome.verdict`, and any per-step field the schema does not declare — a per-step reward is the same signal in credit-assignment form) are moved out of `outcome` and `steps[]` and into `provenance.gated_evidence`, where no config reads them as training input; the exporters re-check the same invariant on every line; and the release build measures the rows it is about to write and refuses to write ANY file if a gated row carries a positive reward — or any positive number derived from one — in ANY config.',\n '',\n 'The table below is **measured on the rows in this release**, not a description of intent — the build counts the emitted rows and this card renders those counts:',\n '',\n markdownTable(\n [\n 'config',\n 'policy',\n 'gated rows written',\n 'gated rows not written',\n 'max reward',\n 'max reward-derived number',\n 'flag',\n ],\n rows,\n ),\n '',\n 'Why gated rows are kept where they are kept: an SFT row is imitated verbatim, so a gamed trajectory must never appear in one at any weight. In `verifiers` the reward is a signed learning signal, and a gamed trajectory at reward 0 is a correct negative — dropping it would bias the negative population toward honest failures and leave a trainer no example of gaming being penalized. `rft` re-samples the completion, so only the prompt and the grader reference ship. `raw` is a faithful audit dump, where the gated row is the one an auditor most wants.',\n '',\n \"Reward 0 is never the only label: every included config carries the flag on the row itself, because zeroing alone makes a faked success indistinguishable from an honest failure. Filter on the flag to drop the gamed population, or select on it to mine it. The gated run's own measurements are not destroyed either — they are parked verbatim under `provenance.gated_evidence` in `raw/train.jsonl`, which is where an auditor can see what the run claimed and why it was flagged.\",\n mixedExclusion\n ? '\\n\"Gated rows not written\" is not all gate: `verifiers` also drops gap lines (empty transcript) and `rft` drops lines with no prompt turn, so that column can mix both causes.'\n : '',\n ]\n .join('\\n')\n .trimEnd()\n}\n\nfunction roleRewardRows(lines: MintedRolloutLine[]): string[][] {\n const counts = new Map<RolloutRole, Map<string, number>>()\n for (const line of lines) {\n const byReward = counts.get(line.role) ?? new Map<string, number>()\n const key = formatReward(line.outcome.reward)\n byReward.set(key, (byReward.get(key) ?? 0) + 1)\n counts.set(line.role, byReward)\n }\n const rows: string[][] = []\n for (const role of ROLLOUT_ROLES) {\n const byReward = counts.get(role)\n if (!byReward) continue\n const keys = [...byReward.keys()].sort((a, b) => {\n if (a === 'null') return 1\n if (b === 'null') return -1\n return Number(b) - Number(a)\n })\n for (const key of keys) rows.push([role, key, String(byReward.get(key))])\n }\n return rows\n}\n\nexport function buildDatasetCard(inputs: DatasetCardInputs): string {\n const {\n lines,\n formats,\n includeProposers,\n sourceFiles,\n scrubTotals,\n excluded,\n formatCounts,\n gate,\n } = inputs\n // The card is the artifact a buyer reads; it may not render a gate report\n // that violates the release policy even when called outside the build.\n assertGateReport(gate)\n\n const runIds = unique(lines.map((line) => line.run_id))\n const generations = [...new Set(lines.map((line) => line.generation))]\n .filter((g): g is number => g !== null)\n .sort((a, b) => a - b)\n const models = unique(lines.map((line) => line.policy.model))\n const harnesses = unique(lines.map((line) => line.policy.harness))\n const captures = unique(lines.map((line) => line.provenance.capture))\n const rewardSources = unique(lines.map((line) => line.outcome.reward_source))\n const gapLines = lines.filter((line) => line.messages.length === 0).length\n const gatedLines = lines.filter((line) => line.outcome.realness_gated).length\n if (gatedLines !== gate.gatedLines) {\n throw new Error(\n `dataset card: gate report claims ${gate.gatedLines} realness-gated line(s) but the lines ` +\n `it describes contain ${gatedLines} — the card and the build measured different data.`,\n )\n }\n\n const configs = formats\n .map((format) =>\n [\n ` - config_name: ${format}`,\n ' data_files:',\n ' - split: train',\n ` path: ${FORMAT_FILES[format]}`,\n ].join('\\n'),\n )\n .join('\\n')\n\n const frontmatter = [\n '---',\n 'license: unknown',\n 'pretty_name: Tangle rollout ledger — agent trajectories',\n 'configs:',\n configs,\n '---',\n ].join('\\n')\n\n const formatsTable = markdownTable(\n ['config', 'path', 'rows', 'contents'],\n formats.map((format) => [\n format,\n `\\`${FORMAT_FILES[format]}\\``,\n String(formatCounts[format] ?? 0),\n FORMAT_DESCRIPTIONS[format],\n ]),\n )\n\n const countsTable = markdownTable(['role', 'reward', 'lines'], roleRewardRows(lines))\n\n const scrubTable = markdownTable(\n ['rule', 'rewrites'],\n Object.entries(scrubTotals).map(([rule, count]) => [rule, String(count)]),\n )\n\n const proposerNote = includeProposers\n ? 'Proposer sessions are INCLUDED (`--include-proposers`); their transcripts contain improvement-loop harness source.'\n : `Proposer sessions are excluded by default (${excluded.proposers} lines dropped); they contain improvement-loop harness source. Rebuild with \\`--include-proposers\\` to keep them.`\n\n return `${frontmatter}\n\n# Tangle rollout ledger — agent trajectories\n\nOne line per agent invocation (supervisor episode, worker session, proposer shot, judge call, analyst pass) captured by the \\`${ROLLOUT_SCHEMA}\\` rollout ledger, labeled with improvement-loop coordinates and the official-judge reward, with the full message transcript inline.\n\nThis release contains the **trainable split only** (\\`search\\`). Holdout, dev, and canary splits are structurally excluded at build time, and the build additionally drops any non-trainable line as a fail-closed filter (${excluded.nonTrain} dropped here).\n\n## Formats\n\n${formatsTable}\n\n## Anti-Goodhart gate (\\`realness_gated\\`)\n\n${gateSection(formats, gate, lines.length)}\n\n## Schema (\\`${ROLLOUT_SCHEMA}\\`)\n\nEach raw line carries:\n\n- \\`rollout_id\\` / \\`parent_rollout_id\\` — invocation identity; workers point at their spawning supervisor episode.\n- \\`run_id\\`, \\`experiment_id\\`, \\`candidate_id\\` — run/experiment/candidate identity from the producing RunRecord, when present.\n- \\`generation\\`, \\`candidate_index\\` — improvement-loop coordinates (\\`-1\\` = baseline campaign; \\`null\\` = not an improvement loop).\n- \\`role\\` — one of ${ROLLOUT_ROLES.map((role) => `\\`${role}\\``).join(', ')}.\n- \\`task\\` — suite, instance id, split, seed, replicate index.\n- \\`policy\\` — harness, model, provider, profile commit, prompt/config hashes, sampling params.\n- \\`messages\\` / \\`tool_defs\\` — full transcript in canonical OpenAI chat-with-tools form (including \\`reasoning_content\\`). An empty \\`messages\\` array is a labeled gap line; \\`provenance.gap\\` says why the transcript could not be recovered (${gapLines} gap lines in this release).\n- \\`outcome\\` — \\`reward\\` (the single scalar), \\`reward_source\\`, the verbatim judge \\`verdict\\`, non-scalar \\`metrics\\`, and \\`realness_gated\\` (anti-Goodhart flag; see [Anti-Goodhart gate](#anti-goodhart-gate-realness_gated) for the per-config treatment and the measured counts).\n- \\`cost\\` — usd, token counts, wall time.\n- \\`artifacts\\` / \\`provenance\\` — patch/run-dir/transcript pointers (scrubbed) and capture metadata.\n\n## Provenance\n\n- Source ledgers: ${sourceFiles.map((file) => `\\`${file}\\``).join(', ')}\n- Run ids: ${runIds.map((id) => `\\`${id}\\``).join(', ')}\n- Generations: ${generations.join(', ')} (\\`-1\\` = baseline campaign)\n- Models: ${models.map((m) => `\\`${m}\\``).join(', ')}\n- Harnesses: ${harnesses.map((h) => `\\`${h}\\``).join(', ')}\n- Capture modes: ${captures.join(', ')}\n- Every reward traces to a named source (reward sources in this release: ${rewardSources.map((s) => `\\`${s}\\``).join(', ')}).\n\n${proposerNote}\n\n## Reward semantics per role\n\n- **agent** — the producing RunRecord's holdout/search score, with the realness gate forcing gamed successes to 0.\n- **supervisor** — the official-judge verdict on the episode's delivered artifact (1 = resolved, 0 = not).\n- **worker** — INHERITED from the parent supervisor episode (\\`…/inherited\\`). Caveat: reward 1 does not establish this worker's individual contribution (sibling workers in the same episode share the episode outcome), and reward 0 does not prove this worker failed.\n- **proposer** — the fraction of improvement-set instances the proposed candidate resolved (\\`…/candidate-resolved-fraction\\`); a scalar in [0, 1], not a binary verdict.\n- **judge / analyst** — carry the episode verdict where one applies; otherwise \\`reward: null\\` (a labeled gap, never 0).\n\n## Counts\n\n${countsTable}\n\nTotal lines: ${lines.length}\n\n## Scrubbing\n\nAbsolute home paths were rewritten to \\`$WORK\\`, credential-shaped strings were replaced with \\`[REDACTED:<kind>]\\` markers, internal hostnames were normalized to \\`*.internal.example\\`, and username-bearing incidentals (\\`ls -l\\` owner columns, per-user pytest tmpdirs) were normalized to \\`user\\` / \\`$USER\\`. Rewrite counts for this release (full per-file breakdown in \\`scrub-report.json\\`):\n\n${scrubTable}\n\n## License\n\n\\`license: unknown\\` is a placeholder — the releasing operator must set the real SPDX license id in the frontmatter above before publishing.\n\n## Citation\n\n\\`\\`\\`bibtex\n@misc{tangle_rollout_ledger,\n title = {Tangle rollout ledger — agent trajectories},\n author = {{Tangle Network}},\n howpublished = {HuggingFace Datasets},\n note = {Operator: fill in the repository URL, authors, and year before publishing}\n}\n\\`\\`\\`\n`\n}\n","/**\n * Deterministic scrubbing pass over rollout-ledger lines before public release.\n *\n * Every rule is a pure regex rewrite applied to every string value in a line\n * (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed\n * line is still a valid `tangle.rollout.v1` line. Rules are idempotent:\n * scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is\n * the property the release pipeline relies on to prove nothing half-scrubbed\n * ships. Rule order matters: whole `KEY=value` env pairs are redacted before\n * the bare-key rule so one secret is never counted twice.\n */\n\nimport { assertMinted, type MintedRolloutLine } from '../schema'\n\nexport interface ScrubRule {\n name: string\n pattern: RegExp\n /** Rewrite for one match; `g1` is the first capture group when present. */\n rewrite: (match: string, g1?: string) => string\n}\n\nconst SCRUB_RULES: readonly ScrubRule[] = [\n {\n // /home/<user> and /Users/<user> prefixes → $WORK ($HOME-derived paths).\n name: 'home-path',\n pattern: /\\/(?:home|Users)\\/[A-Za-z0-9._-]+/g,\n rewrite: () => '$WORK',\n },\n {\n // Harness stores encode cwd as a dashed path segment (-home-drew-…);\n // strip the leading -home-<user> so the username never ships.\n name: 'home-path-encoded',\n pattern: /(?<=\\/)-(?:home|Users)-[A-Za-z0-9_.]+/g,\n rewrite: () => '$WORK',\n },\n {\n // pytest names its tmpdir after the invoking user (/tmp/pytest-of-drew).\n name: 'tmp-user-dir',\n pattern: /\\/tmp\\/pytest-of-[A-Za-z0-9._-]+/g,\n rewrite: () => '/tmp/pytest-of-$USER',\n },\n {\n // `ls -l` owner/group columns leak the username in captured tool output.\n // The `(?!user user…)` guard keeps a second pass at zero rewrites.\n name: 'ls-owner',\n pattern:\n /([-bcdlps][-rwxsStT]{9}[.+@]?\\s+\\d+\\s+)(?!user user(?=\\s))[A-Za-z0-9._-]+\\s+[A-Za-z0-9._-]+(?=\\s)/g,\n rewrite: (_match, prefix) => `${prefix}user user`,\n },\n {\n // Env-var-shaped secrets: NAME=value where NAME looks credential-bearing.\n // The (?!\\[REDACTED:) guard keeps the rule idempotent on its own output.\n name: 'env-secret',\n pattern:\n /\\b([A-Z][A-Z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIALS?))=(?!\\[REDACTED:)(\"[^\"]*\"|'[^']*'|[^\\s\"']+)/g,\n rewrite: (_match, name) => `${name}=[REDACTED:env]`,\n },\n {\n name: 'bearer-token',\n pattern: /\\b(Bearer|Basic)\\s+(?!\\[REDACTED:)[A-Za-z0-9\\-._~+/=]{8,}/g,\n rewrite: (_match, scheme) => `${scheme} [REDACTED:bearer]`,\n },\n {\n // Bare provider-prefixed keys (OpenAI/Anthropic sk-, GitHub ghp_/gho_/…,\n // fine-grained PATs, HuggingFace hf_, Slack xox*, AWS AKIA).\n name: 'api-key',\n pattern:\n /\\b(?:sk-[A-Za-z0-9_-]{16,}|gh[pousr]_[A-Za-z0-9]{16,}|github_pat_[A-Za-z0-9_]{20,}|hf_[A-Za-z0-9]{16,}|xox[baprs]-[A-Za-z0-9-]{10,}|AKIA[0-9A-Z]{16})\\b/g,\n rewrite: () => '[REDACTED:api-key]',\n },\n {\n // Our infra hostnames → placeholder domain, subdomain preserved\n // (router.tangle.tools → router.internal.example).\n name: 'infra-host',\n pattern: /(?<![A-Za-z0-9.-])((?:[A-Za-z0-9-]+\\.)*)tangle\\.(?:tools|network)(?![A-Za-z0-9-])/g,\n rewrite: (_match, prefix) => `${prefix ?? ''}internal.example`,\n },\n {\n // Known workstation hostnames leak through email Message-IDs and fqdn\n // lookups inside captured test output; extend the list as machines join.\n name: 'machine-host',\n pattern: /\\b[A-Za-z0-9]+-GTR-Pro\\b/g,\n rewrite: () => 'workstation',\n },\n]\n\n/** Rule name → number of matches rewritten. Always carries every rule (0 is data). */\nexport type ScrubCounts = Record<string, number>\n\nexport function emptyScrubCounts(): ScrubCounts {\n return Object.fromEntries(SCRUB_RULES.map((rule) => [rule.name, 0]))\n}\n\nexport function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts {\n for (const [name, count] of Object.entries(from)) into[name] = (into[name] ?? 0) + count\n return into\n}\n\nexport function scrubText(text: string, counts: ScrubCounts): string {\n let out = text\n for (const rule of SCRUB_RULES) {\n out = out.replace(rule.pattern, (match: string, ...rest: unknown[]) => {\n counts[rule.name] = (counts[rule.name] ?? 0) + 1\n return rule.rewrite(match, typeof rest[0] === 'string' ? rest[0] : undefined)\n })\n }\n return out\n}\n\nfunction scrubValue(value: unknown, counts: ScrubCounts): unknown {\n if (typeof value === 'string') return scrubText(value, counts)\n if (Array.isArray(value)) return value.map((item) => scrubValue(item, counts))\n if (value !== null && typeof value === 'object') {\n const out: Record<string, unknown> = {}\n for (const [key, item] of Object.entries(value)) out[key] = scrubValue(item, counts)\n return out\n }\n return value\n}\n\n/**\n * Scrub every string value in a line; structure and key order are preserved.\n *\n * `assertMinted` on the way out rather than a cast: scrubbing rebuilds the\n * object, so the brand has to be re-earned, and re-validating proves the rules\n * did not rewrite a field the schema constrains (`reward` is a number, not a\n * string, so no rule should ever touch it — this is what checks that).\n */\nexport function scrubRolloutLine(line: MintedRolloutLine, counts: ScrubCounts): MintedRolloutLine {\n return assertMinted(scrubValue(line, counts), `scrubbed rollout line ${line.rollout_id}`)\n}\n\nexport function scrubLines(lines: MintedRolloutLine[]): {\n lines: MintedRolloutLine[]\n counts: ScrubCounts\n} {\n const counts = emptyScrubCounts()\n return { lines: lines.map((line) => scrubRolloutLine(line, counts)), counts }\n}\n\n/**\n * A `RolloutScrubber` (text → text) applying the full rule set — the\n * default hook to pass to `mintRolloutRows({ scrub })` so lines are\n * scrubbed at mint time, before they ever reach a ledger file. Release\n * builds re-run `scrubLines` regardless (idempotent), so double-scrubbing\n * is safe and counted as zero.\n */\nexport function defaultRolloutScrubber(text: string): string {\n return scrubText(text, emptyScrubCounts())\n}\n","/**\n * One-command HuggingFace dataset release from rollout ledgers:\n *\n * agent-eval rollout-release <ledger.jsonl...> --out <dir> \\\n * [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]\n *\n * Pipeline per input ledger: read + validate → fail-closed filters\n * (trainable split only; proposer sessions dropped unless\n * --include-proposers, they contain improvement-loop harness source) →\n * deterministic scrub → export the requested formats + scrub-report.json +\n * auto-generated README.md card. Deterministic: same inputs and flags →\n * byte-identical output dir.\n *\n * --push uploads the built dir with `huggingface-cli upload` only when the\n * CLI exists on PATH and HF_TOKEN is present in the env; the token is\n * never printed. Everything else runs fully offline.\n */\n\nimport { spawnSync } from 'node:child_process'\nimport { mkdir, writeFile } from 'node:fs/promises'\nimport { basename, dirname, join } from 'node:path'\nimport { toJsonl, toRftItems, toSftRows, toVerifiersRolloutOutputs } from '../exporters'\nimport { readRolloutLedger, writeRolloutLedger } from '../ledger'\nimport { isTrainableSplit, type MintedRolloutLine } from '../schema'\nimport { buildDatasetCard, FORMAT_FILES, RELEASE_FORMATS, type ReleaseFormat } from './card'\nimport {\n assertGateReport,\n FORMAT_GATE_DISPOSITION,\n type GateReport,\n gatedRolloutIds,\n measureFormatGate,\n releaseRowRefs,\n} from './gate-report'\nimport { addScrubCounts, emptyScrubCounts, type ScrubCounts, scrubLines } from './scrub'\n\nexport interface BuildOptions {\n out: string\n formats: ReleaseFormat[]\n includeProposers: boolean\n}\n\nexport interface ScrubReport {\n /** Input ledger path → rule → rewrite count (only shipped lines are scrubbed). */\n files: Record<string, ScrubCounts>\n totals: ScrubCounts\n excluded: { proposers: number; nonTrain: number }\n}\n\nexport interface BuildSummary {\n inputs: string[]\n read: number\n kept: number\n scrub: ScrubReport\n formatCounts: Partial<Record<ReleaseFormat, number>>\n /** Per-format anti-Goodhart accounting, measured on the rows written. */\n gate: GateReport\n files: string[]\n}\n\nexport async function buildHfDataset(\n inputs: string[],\n options: BuildOptions,\n): Promise<BuildSummary> {\n if (inputs.length === 0) throw new Error('no input ledgers given')\n if (options.formats.length === 0) throw new Error('no formats selected')\n\n const report: ScrubReport = {\n files: {},\n totals: emptyScrubCounts(),\n excluded: { proposers: 0, nonTrain: 0 },\n }\n const kept: MintedRolloutLine[] = []\n let read = 0\n\n for (const input of inputs) {\n const lines = await readRolloutLedger(input)\n read += lines.length\n const shippable = lines.filter((line) => {\n if (!isTrainableSplit(line.task.split)) {\n report.excluded.nonTrain += 1\n return false\n }\n if (!options.includeProposers && line.role === 'proposer') {\n report.excluded.proposers += 1\n return false\n }\n return true\n })\n const scrubbed = scrubLines(shippable)\n report.files[input] = scrubbed.counts\n addScrubCounts(report.totals, scrubbed.counts)\n kept.push(...scrubbed.lines)\n }\n\n const formatCounts: Partial<Record<ReleaseFormat, number>> = {}\n const files: string[] = []\n const gated = gatedRolloutIds(kept)\n const gate: GateReport = { gatedLines: gated.size, byFormat: {} }\n\n // Every selected format's rows are exported and gate-measured BEFORE the\n // first byte is written. A build that would ship a gamed run at a positive\n // reward fails with nothing on disk, rather than leaving a poisoned config\n // behind for someone to `--push`.\n const pending: Array<{ path: string; write: () => Promise<void> }> = []\n\n for (const format of options.formats) {\n const path = join(options.out, FORMAT_FILES[format])\n if (format === 'raw') {\n // writeRolloutLedger re-validates every scrubbed line before it lands.\n gate.byFormat.raw = measureFormatGate(gated, releaseRowRefs.raw(kept))\n formatCounts.raw = kept.length\n pending.push({ path, write: () => writeRolloutLedger(path, kept) })\n } else if (format === 'sft') {\n const rows = toSftRows(kept)\n gate.byFormat.sft = measureFormatGate(gated, releaseRowRefs.sft(rows))\n formatCounts.sft = rows.length\n pending.push({ path, write: () => writeFile(path, toJsonl(rows)) })\n } else if (format === 'verifiers') {\n // The declared disposition drives the exporter: 'zero-and-flag' ships\n // gated lines and honest failures as labeled negatives at reward 0.\n const outputs = toVerifiersRolloutOutputs(kept, {\n gatedLines: FORMAT_GATE_DISPOSITION.verifiers,\n })\n gate.byFormat.verifiers = measureFormatGate(gated, releaseRowRefs.verifiers(outputs))\n formatCounts.verifiers = outputs.length\n pending.push({ path, write: () => writeFile(path, toJsonl(outputs)) })\n } else {\n const items = toRftItems(kept, { gatedLines: FORMAT_GATE_DISPOSITION.rft })\n gate.byFormat.rft = measureFormatGate(gated, releaseRowRefs.rft(items))\n formatCounts.rft = items.length\n pending.push({ path, write: () => writeFile(path, toJsonl(items)) })\n }\n }\n\n assertGateReport(gate)\n\n for (const { path, write } of pending) {\n await mkdir(dirname(path), { recursive: true })\n await write()\n files.push(path)\n }\n\n const reportPath = join(options.out, 'scrub-report.json')\n await writeFile(reportPath, `${JSON.stringify(report, null, 2)}\\n`)\n files.push(reportPath)\n\n const cardPath = join(options.out, 'README.md')\n await writeFile(\n cardPath,\n buildDatasetCard({\n lines: kept,\n formats: options.formats,\n includeProposers: options.includeProposers,\n sourceFiles: inputs.map((input) => basename(input)),\n scrubTotals: report.totals,\n excluded: report.excluded,\n formatCounts,\n gate,\n }),\n )\n files.push(cardPath)\n\n return { inputs, read, kept: kept.length, scrub: report, formatCounts, gate, files }\n}\n\nexport function planPushCommand(repo: string, outDir: string): string[] {\n return ['huggingface-cli', 'upload', repo, outDir, '.', '--repo-type', 'dataset']\n}\n\nfunction pushDataset(repo: string, outDir: string): void {\n const found = spawnSync('which', ['huggingface-cli'], { stdio: 'ignore' })\n if (found.status !== 0) {\n throw new Error(\n 'huggingface-cli not found on PATH — install huggingface_hub[cli] before --push',\n )\n }\n if (!process.env.HF_TOKEN) {\n throw new Error('HF_TOKEN not present in env — refusing to push')\n }\n const [command, ...args] = planPushCommand(repo, outDir) as [string, ...string[]]\n // Token stays in the inherited env; it is never echoed or interpolated.\n const run = spawnSync(command, args, { stdio: 'inherit' })\n if (run.status !== 0) throw new Error(`huggingface-cli upload exited ${String(run.status)}`)\n}\n\nexport interface RolloutReleaseCliArgs extends BuildOptions {\n inputs: string[]\n push: string | null\n}\n\nconst ROLLOUT_RELEASE_USAGE =\n 'usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]'\n\nexport function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs {\n const args: RolloutReleaseCliArgs = {\n inputs: [],\n out: '',\n formats: [...RELEASE_FORMATS],\n includeProposers: false,\n push: null,\n }\n for (let i = 0; i < argv.length; i++) {\n const arg = argv[i]!\n if (arg === '--out') {\n args.out = argv[++i] ?? ''\n } else if (arg === '--formats') {\n const raw = (argv[++i] ?? '').split(',').filter(Boolean)\n for (const format of raw) {\n if (!RELEASE_FORMATS.includes(format as ReleaseFormat)) {\n throw new Error(\n `unknown format \"${format}\" — expected one of ${RELEASE_FORMATS.join(',')}`,\n )\n }\n }\n args.formats = raw as ReleaseFormat[]\n } else if (arg === '--include-proposers') {\n args.includeProposers = true\n } else if (arg === '--push') {\n args.push = argv[++i] ?? null\n } else if (arg.startsWith('--')) {\n throw new Error(`unknown flag \"${arg}\"`)\n } else {\n args.inputs.push(arg)\n }\n }\n if (args.inputs.length === 0 || !args.out) {\n throw new Error(ROLLOUT_RELEASE_USAGE)\n }\n if (args.push !== null && !/^[\\w.-]+\\/[\\w.-]+$/.test(args.push)) {\n throw new Error(`--push expects <org/name>, got \"${args.push}\"`)\n }\n return args\n}\n\n/** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */\nexport async function runRolloutReleaseCli(argv: string[]): Promise<number> {\n let args: RolloutReleaseCliArgs\n try {\n args = parseRolloutReleaseArgs(argv)\n } catch (error) {\n process.stderr.write(`${error instanceof Error ? error.message : String(error)}\\n`)\n return 2\n }\n const summary = await buildHfDataset(args.inputs, args)\n process.stdout.write(\n `${JSON.stringify(\n {\n read: summary.read,\n kept: summary.kept,\n formatCounts: summary.formatCounts,\n gate: summary.gate,\n scrub: summary.scrub,\n },\n null,\n 2,\n )}\\n`,\n )\n process.stdout.write(`dataset → ${args.out} (${summary.files.length} files)\\n`)\n if (args.push !== null) {\n pushDataset(args.push, args.out)\n process.stdout.write(`pushed → ${args.push}\\n`)\n }\n return 0\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;AAoBA,SAAS,UAAU,OAA8B;CAC/C,KAAK,MAAM,CAAC,GAAG,SAAS,MAAM,QAAQ,GAAG,kBAAkB,MAAM,iBAAiB,EAAE,EAAE;CACtF,OAAO,MAAM,KAAK,SAAS,KAAK,UAAU,IAAI,CAAC,CAAC,CAAC,KAAK,IAAI,KAAK,MAAM,SAAS,IAAI,OAAO;AAC3F;;AAGA,eAAsB,mBAAmB,MAAc,OAAqC;CAC1F,MAAM,UAAU,UAAU,KAAK;CAC/B,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAC9C,MAAM,UAAU,MAAM,OAAO;AAC/B;;AAGA,eAAsB,mBAAmB,MAAc,OAAqC;CAC1F,IAAI,MAAM,WAAW,GAAG;CACxB,MAAM,UAAU,UAAU,KAAK;CAC/B,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAC9C,MAAM,WAAW,MAAM,OAAO;AAChC;;;;;;;;;;;AAYA,eAAsB,kBAAkB,MAA4C;CAClF,OAAO,UAAU,OAAO,QAAQ,YAAY,aAAa,QAAQ,OAAO,CAAC;AAC3E;;;;;;;;;;;;;;;AAgBA,eAAsB,mBAAmB,MAAsC;CAC7E,OAAO,UAAU,OAAO,QAAQ,YAAyB;EACvD,kBAAkB,QAAQ,OAAO;EACjC,OAAO;CACT,CAAC;AACH;AAEA,eAAe,UACb,MACA,OACc;CACd,MAAM,MAAM,MAAM,SAAS,MAAM,MAAM;CACvC,MAAM,QAAa,CAAC;CACpB,MAAM,WAAW,IAAI,MAAM,IAAI;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,SAAS,QAAQ,KAAK;EACxC,MAAM,OAAO,SAAS;EACtB,IAAI,CAAC,MAAM,KAAK,GAAG;EACnB,IAAI;EACJ,IAAI;GACF,SAAS,KAAK,MAAM,IAAI;EAC1B,SAAS,OAAO;GACd,MAAM,IAAI,MACR,GAAG,KAAK,GAAG,IAAI,EAAE,qBAAqB,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,GAC7F;EACF;EACA,MAAM,KAAK,MAAM,QAAQ,GAAG,KAAK,GAAG,IAAI,GAAG,CAAC;CAC9C;CACA,OAAO;AACT;;;ACtCA,MAAa,0BAAkE;CAC7E,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;;AA2GA,SAAgB,gBAAgB,OAAkD;CAChF,OAAO,IAAI,IAAI,MAAM,QAAQ,SAAS,KAAK,QAAQ,cAAc,CAAC,CAAC,KAAK,SAAS,KAAK,UAAU,CAAC;AACnG;;;;;;;AAQA,MAAa,iBAAiB;CAI5B,MAAM,SACJ,KAAK,KAAK,SAAS;EACjB,YAAY,IAAI,SAAS;EACzB,QAAQ,IAAI,SAAS;EACrB,mBAAmB,IAAI,SAAS;CAClC,EAAE;CAIJ,YAAY,SACV,KAAK,KAAK,SAAS;EACjB,YAAY,IAAI,KAAK;EACrB,QAAQ,IAAI;EACZ,UAAU,IAAI;EACd,mBAAmB,IAAI,KAAK;CAC9B,EAAE;CAGJ,MAAM,SACJ,KAAK,KAAK,SAAS;EACjB,YAAY,IAAI,UAAU;EAC1B,QAAQ,IAAI,UAAU;EACtB,UAAU,IAAI,UAAU;EACxB,mBAAmB,IAAI,UAAU;CACnC,EAAE;CACJ,MAAM,UACJ,MAAM,KAAK,UAAU;EACnB,YAAY,KAAK;EACjB,QAAQ,KAAK,QAAQ;EAIrB,UAAU;GAAE,SAAS,KAAK,QAAQ;GAAS,SAAS,KAAK,QAAQ;EAAQ;EAGzE,cAAc,sBAAsB,KAAK,KAAK;EAC9C,mBAAmB,KAAK,QAAQ,qBAAqB;CACvD,EAAE;AACN;;AAGA,SAAS,kBAAkB,OAAgB,MAAiC;CAC1E,IAAI,OAAO,UAAU,UACnB,OAAO,OAAO,SAAS,KAAK,KAAK,QAAQ,IAAI,CAAC;EAAE;EAAM;CAAM,CAAC,IAAI,CAAC;CAEpE,IAAI,MAAM,QAAQ,KAAK,GACrB,OAAO,MAAM,SAAS,MAAM,MAAM,kBAAkB,MAAM,GAAG,KAAK,GAAG,EAAE,EAAE,CAAC;CAE5E,IAAI,OAAO,UAAU,YAAY,UAAU,MACzC,OAAO,OAAO,QAAQ,KAAK,CAAC,CAAC,SAAS,CAAC,KAAK,WAC1C,kBAAkB,OAAO,SAAS,KAAK,MAAM,GAAG,KAAK,GAAG,KAAK,CAC/D;CAEF,OAAO,CAAC;AACV;;AAGA,SAAgB,kBACd,OACA,MACkB;CAClB,MAAM,UAAU,KAAK,QAAQ,QAAQ,MAAM,IAAI,IAAI,UAAU,CAAC;CAC9D,MAAM,UAAU,QAAQ,KAAK,QAAQ,IAAI,MAAM,CAAC,CAAC,QAAQ,MAAmB,MAAM,IAAI;CACtF,MAAM,WAAW,QAAQ,SAAS,QAAQ,kBAAkB,IAAI,UAAU,EAAE,CAAC;CAC7E,MAAM,eAAe,QAAQ,SAAS,QAAQ,kBAAkB,IAAI,cAAc,OAAO,CAAC;CAC1F,MAAM,aAAa,KAChB,QAAQ,QAAQ,IAAI,sBAAsB,KAAK,CAAC,CAChD,KAAK,QAAQ,IAAI,MAAM,CAAC,CACxB,QAAQ,MAAmB,MAAM,QAAQ,IAAI,CAAC;CACjD,OAAO;EACL,OAAO,MAAM;EACb,SAAS,QAAQ;EACjB,UAAU,MAAM,OAAO,QAAQ;EAC/B,kBAAkB,QAAQ,WAAW,IAAI,OAAO,KAAK,IAAI,GAAG,OAAO;EACnE,oBAAoB,SAAS,QAC1B,MAAM,UAAW,SAAS,QAAQ,MAAM,QAAQ,KAAK,QAAQ,QAAQ,MACtE,IACF;EACA,wBAAwB,WAAW;EACnC,qBAAqB,WAAW,WAAW,IAAI,OAAO,KAAK,IAAI,GAAG,UAAU;EAC5E,wBAAwB,aAAa,QAClC,MAAM,UAAW,SAAS,QAAQ,MAAM,QAAQ,KAAK,QAAQ,QAAQ,MACtE,IACF;CACF;AACF;;;;;;;;;;;;;AAcA,MAAM,kBAEF;CACF,wBAAwB,QAAQ,WAC9B,OAAO,qBAAqB,QAAQ,OAAO,oBAAoB,IAC3D,OACA,mBAAmB,OAAO,KAAK,OAAO,QAAQ,sDAC/B,OAAO,iBAAiB;CAE7C,mBAAmB,QAAQ,WAAW;EACpC,IAAI,OAAO,uBAAuB,MAAM,OAAO;EAC/C,MAAM,EAAE,MAAM,UAAU,OAAO;EAC/B,OACE,mBAAmB,OAAO,KAAK,OAAO,QAAQ,iEACpB,KAAK,KAAK,MAAM;CAK9C;CACA,4BAA4B,QAAQ,WAAW;EAC7C,IAAI,OAAO,2BAA2B,MAAM,OAAO;EACnD,MAAM,EAAE,MAAM,UAAU,OAAO;EAC/B,OACE,mBAAmB,OAAO,KAAK,OAAO,QAAQ,gHAC2B,KAAK,KAAK,MAAM;CAK7F;CACA,sBAAsB,QAAQ,WAC5B,OAAO,wBAAwB,OAC3B,OACA,mBAAmB,OAAO,KAAK,OAAO,uBAAuB,uCAC9C,OAAO,oBAAoB;AAIlD;;;;;;;;;;;;;;;;AAiBA,SAAgB,iBAAiB,QAA0B;CACzD,KAAK,MAAM,CAAC,QAAQ,WAAW,OAAO,QAAQ,OAAO,QAAQ,GAE1D;EACD,KAAK,MAAM,MAAM,gBAAgB;GAC/B,IAAI,cAAc,iBAAiB,GAAG,CAAC,SAAS,WAAW;GAC3D,MAAM,UAAU,gBAAgB,GAAG,CAAC,QAAQ,MAAM;GAClD,IAAI,YAAY,MAAM,MAAM,IAAI,MAAM,OAAO;EAC/C;EAIA,IAAI,wBAAwB,YAAY,aAAa,OAAO,UAAU,GACpE,MAAM,IAAI,MACR,mBAAmB,OAAO,4CAA4C,OAAO,QAAQ,UAEvF;CAEJ;AACF;;;;;;;;;;;;ACxVA,MAAa,kBAAkB;CAAC;CAAO;CAAa;CAAO;AAAK;;AAIhE,MAAa,eAA8C;CACzD,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;AAEA,MAAM,sBAAqD;CACzD,KAAK;CACL,WACE;CACF,KAAK;CACL,KAAK,UAAU,eAAe;AAChC;AAqBA,SAAS,OAAO,QAAwC;CACtD,OAAO,CAAC,GAAG,IAAI,IAAI,OAAO,QAAQ,MAAmB,MAAM,IAAI,CAAC,CAAC,CAAC,CAAC,KAAK;AAC1E;AAEA,SAAS,aAAa,QAA+B;CACnD,IAAI,WAAW,MAAM,OAAO;CAC5B,OAAO,OAAO,UAAU,MAAM,IAAI,OAAO,MAAM,IAAI,OAAO,QAAQ,CAAC;AACrE;AAEA,SAAS,cAAc,QAAkB,MAA0B;CACjE,OAAO;EACL,KAAK,OAAO,KAAK,KAAK,EAAE;EACxB,KAAK,OAAO,UAAU,KAAK,CAAC,CAAC,KAAK,KAAK,EAAE;EACzC,GAAG,KAAK,KAAK,QAAQ,KAAK,IAAI,KAAK,KAAK,EAAE,GAAG;CAC/C,CAAC,CAAC,KAAK,IAAI;AACb;;AAGA,MAAM,kBAAiD;CACrD,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;AAEA,MAAM,oBAAmD;CACvD,KAAK;CACL,WAAW;CACX,KAAK;CACL,KAAK;AACP;AAEA,SAAS,YAAY,SAA0B,MAAkB,YAA4B;CAC3F,MAAM,OAAO,QAAQ,KAAK,WAAW;EACnC,MAAM,SAAS,KAAK,SAAS;EAC7B,OAAO;GACL;GACA,kBAAkB;GAClB,OAAO,QAAQ,WAAW,CAAC;GAC3B,OAAO,QAAQ,YAAY,CAAC;GAC5B,QAAQ,qBAAqB,QAAQ,WAAW,KAAA,IAC5C,MACA,aAAa,OAAO,gBAAgB;GACxC,QAAQ,sBAAsB,OAC1B,MACA,GAAG,OAAO,mBAAmB,KAAK,KAAK,OAAO,mBAAmB;GACrE,gBAAgB;EAClB;CACF,CAAC;CACD,MAAM,iBAAiB,QAAQ,MAC5B,WAAW,wBAAwB,YAAY,eAClD;CACA,OAAO;EACL,kIAAkI,KAAK,WAAW,MAAM,WAAW;EACnK;EACA;EACA;EACA;EACA;EACA,cACE;GACE;GACA;GACA;GACA;GACA;GACA;GACA;EACF,GACA,IACF;EACA;EACA;EACA;EACA;EACA,iBACI,qLACA;CACN,CAAC,CACE,KAAK,IAAI,CAAC,CACV,QAAQ;AACb;AAEA,SAAS,eAAe,OAAwC;CAC9D,MAAM,yBAAS,IAAI,IAAsC;CACzD,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,WAAW,OAAO,IAAI,KAAK,IAAI,qBAAK,IAAI,IAAoB;EAClE,MAAM,MAAM,aAAa,KAAK,QAAQ,MAAM;EAC5C,SAAS,IAAI,MAAM,SAAS,IAAI,GAAG,KAAK,KAAK,CAAC;EAC9C,OAAO,IAAI,KAAK,MAAM,QAAQ;CAChC;CACA,MAAM,OAAmB,CAAC;CAC1B,KAAK,MAAM,QAAQ,eAAe;EAChC,MAAM,WAAW,OAAO,IAAI,IAAI;EAChC,IAAI,CAAC,UAAU;EACf,MAAM,OAAO,CAAC,GAAG,SAAS,KAAK,CAAC,CAAC,CAAC,MAAM,GAAG,MAAM;GAC/C,IAAI,MAAM,QAAQ,OAAO;GACzB,IAAI,MAAM,QAAQ,OAAO;GACzB,OAAO,OAAO,CAAC,IAAI,OAAO,CAAC;EAC7B,CAAC;EACD,KAAK,MAAM,OAAO,MAAM,KAAK,KAAK;GAAC;GAAM;GAAK,OAAO,SAAS,IAAI,GAAG,CAAC;EAAC,CAAC;CAC1E;CACA,OAAO;AACT;AAEA,SAAgB,iBAAiB,QAAmC;CAClE,MAAM,EACJ,OACA,SACA,kBACA,aACA,aACA,UACA,cACA,SACE;CAGJ,iBAAiB,IAAI;CAErB,MAAM,SAAS,OAAO,MAAM,KAAK,SAAS,KAAK,MAAM,CAAC;CACtD,MAAM,cAAc,CAAC,GAAG,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,UAAU,CAAC,CAAC,CAAC,CACnE,QAAQ,MAAmB,MAAM,IAAI,CAAC,CACtC,MAAM,GAAG,MAAM,IAAI,CAAC;CACvB,MAAM,SAAS,OAAO,MAAM,KAAK,SAAS,KAAK,OAAO,KAAK,CAAC;CAC5D,MAAM,YAAY,OAAO,MAAM,KAAK,SAAS,KAAK,OAAO,OAAO,CAAC;CACjE,MAAM,WAAW,OAAO,MAAM,KAAK,SAAS,KAAK,WAAW,OAAO,CAAC;CACpE,MAAM,gBAAgB,OAAO,MAAM,KAAK,SAAS,KAAK,QAAQ,aAAa,CAAC;CAC5E,MAAM,WAAW,MAAM,QAAQ,SAAS,KAAK,SAAS,WAAW,CAAC,CAAC,CAAC;CACpE,MAAM,aAAa,MAAM,QAAQ,SAAS,KAAK,QAAQ,cAAc,CAAC,CAAC;CACvE,IAAI,eAAe,KAAK,YACtB,MAAM,IAAI,MACR,oCAAoC,KAAK,WAAW,6DAC1B,WAAW,mDACvC;CAcF,MAAM,cAAc;EAClB;EACA;EACA;EACA;EAfc,QACb,KAAK,WACJ;GACE,oBAAoB;GACpB;GACA;GACA,iBAAiB,aAAa;EAChC,CAAC,CAAC,KAAK,IAAI,CACb,CAAC,CACA,KAAK,IAOA;EACN;CACF,CAAC,CAAC,KAAK,IAAI;CAEX,MAAM,eAAe,cACnB;EAAC;EAAU;EAAQ;EAAQ;CAAU,GACrC,QAAQ,KAAK,WAAW;EACtB;EACA,KAAK,aAAa,QAAQ;EAC1B,OAAO,aAAa,WAAW,CAAC;EAChC,oBAAoB;CACtB,CAAC,CACH;CAEA,MAAM,cAAc,cAAc;EAAC;EAAQ;EAAU;CAAO,GAAG,eAAe,KAAK,CAAC;CAEpF,MAAM,aAAa,cACjB,CAAC,QAAQ,UAAU,GACnB,OAAO,QAAQ,WAAW,CAAC,CAAC,KAAK,CAAC,MAAM,WAAW,CAAC,MAAM,OAAO,KAAK,CAAC,CAAC,CAC1E;CAEA,MAAM,eAAe,mBACjB,uHACA,8CAA8C,SAAS,UAAU;CAErE,OAAO,GAAG,YAAY;;;;gIAIwG,eAAe;;6NAE8E,SAAS,SAAS;;;;EAI7O,aAAa;;;;EAIb,YAAY,SAAS,MAAM,MAAM,MAAM,EAAE;;eAE5B,eAAe;;;;;;;sBAOR,cAAc,KAAK,SAAS,KAAK,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;;;qPAGyK,SAAS;;;;;;;oBAO1O,YAAY,KAAK,SAAS,KAAK,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;aAC3D,OAAO,KAAK,OAAO,KAAK,GAAG,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;iBACvC,YAAY,KAAK,IAAI,EAAE;YAC5B,OAAO,KAAK,MAAM,KAAK,EAAE,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;eACtC,UAAU,KAAK,MAAM,KAAK,EAAE,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;mBACxC,SAAS,KAAK,IAAI,EAAE;2EACoC,cAAc,KAAK,MAAM,KAAK,EAAE,GAAG,CAAC,CAAC,KAAK,IAAI,EAAE;;EAEzH,aAAa;;;;;;;;;;;;EAYb,YAAY;;eAEC,MAAM,OAAO;;;;;;EAM1B,WAAW;;;;;;;;;;;;;;;;;AAiBb;;;;;;;;;;;;;;AC/RA,MAAM,cAAoC;CACxC;EAEE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;CACA;EAGE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;CACA;EAEE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;CACA;EAGE,MAAM;EACN,SACE;EACF,UAAU,QAAQ,WAAW,GAAG,OAAO;CACzC;CACA;EAGE,MAAM;EACN,SACE;EACF,UAAU,QAAQ,SAAS,GAAG,KAAK;CACrC;CACA;EACE,MAAM;EACN,SAAS;EACT,UAAU,QAAQ,WAAW,GAAG,OAAO;CACzC;CACA;EAGE,MAAM;EACN,SACE;EACF,eAAe;CACjB;CACA;EAGE,MAAM;EACN,SAAS;EACT,UAAU,QAAQ,WAAW,GAAG,UAAU,GAAG;CAC/C;CACA;EAGE,MAAM;EACN,SAAS;EACT,eAAe;CACjB;AACF;AAKA,SAAgB,mBAAgC;CAC9C,OAAO,OAAO,YAAY,YAAY,KAAK,SAAS,CAAC,KAAK,MAAM,CAAC,CAAC,CAAC;AACrE;AAEA,SAAgB,eAAe,MAAmB,MAAgC;CAChF,KAAK,MAAM,CAAC,MAAM,UAAU,OAAO,QAAQ,IAAI,GAAG,KAAK,SAAS,KAAK,SAAS,KAAK;CACnF,OAAO;AACT;AAEA,SAAgB,UAAU,MAAc,QAA6B;CACnE,IAAI,MAAM;CACV,KAAK,MAAM,QAAQ,aACjB,MAAM,IAAI,QAAQ,KAAK,UAAU,OAAe,GAAG,SAAoB;EACrE,OAAO,KAAK,SAAS,OAAO,KAAK,SAAS,KAAK;EAC/C,OAAO,KAAK,QAAQ,OAAO,OAAO,KAAK,OAAO,WAAW,KAAK,KAAK,KAAA,CAAS;CAC9E,CAAC;CAEH,OAAO;AACT;AAEA,SAAS,WAAW,OAAgB,QAA8B;CAChE,IAAI,OAAO,UAAU,UAAU,OAAO,UAAU,OAAO,MAAM;CAC7D,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO,MAAM,KAAK,SAAS,WAAW,MAAM,MAAM,CAAC;CAC7E,IAAI,UAAU,QAAQ,OAAO,UAAU,UAAU;EAC/C,MAAM,MAA+B,CAAC;EACtC,KAAK,MAAM,CAAC,KAAK,SAAS,OAAO,QAAQ,KAAK,GAAG,IAAI,OAAO,WAAW,MAAM,MAAM;EACnF,OAAO;CACT;CACA,OAAO;AACT;;;;;;;;;AAUA,SAAgB,iBAAiB,MAAyB,QAAwC;CAChG,OAAO,aAAa,WAAW,MAAM,MAAM,GAAG,yBAAyB,KAAK,YAAY;AAC1F;AAEA,SAAgB,WAAW,OAGzB;CACA,MAAM,SAAS,iBAAiB;CAChC,OAAO;EAAE,OAAO,MAAM,KAAK,SAAS,iBAAiB,MAAM,MAAM,CAAC;EAAG;CAAO;AAC9E;;;;;;;;AASA,SAAgB,uBAAuB,MAAsB;CAC3D,OAAO,UAAU,MAAM,iBAAiB,CAAC;AAC3C;;;;;;;;;;;;;;;;;;;;AC1FA,eAAsB,eACpB,QACA,SACuB;CACvB,IAAI,OAAO,WAAW,GAAG,MAAM,IAAI,MAAM,wBAAwB;CACjE,IAAI,QAAQ,QAAQ,WAAW,GAAG,MAAM,IAAI,MAAM,qBAAqB;CAEvE,MAAM,SAAsB;EAC1B,OAAO,CAAC;EACR,QAAQ,iBAAiB;EACzB,UAAU;GAAE,WAAW;GAAG,UAAU;EAAE;CACxC;CACA,MAAM,OAA4B,CAAC;CACnC,IAAI,OAAO;CAEX,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,QAAQ,MAAM,kBAAkB,KAAK;EAC3C,QAAQ,MAAM;EAYd,MAAM,WAAW,WAXC,MAAM,QAAQ,SAAS;GACvC,IAAI,CAAC,iBAAiB,KAAK,KAAK,KAAK,GAAG;IACtC,OAAO,SAAS,YAAY;IAC5B,OAAO;GACT;GACA,IAAI,CAAC,QAAQ,oBAAoB,KAAK,SAAS,YAAY;IACzD,OAAO,SAAS,aAAa;IAC7B,OAAO;GACT;GACA,OAAO;EACT,CACoC,CAAC;EACrC,OAAO,MAAM,SAAS,SAAS;EAC/B,eAAe,OAAO,QAAQ,SAAS,MAAM;EAC7C,KAAK,KAAK,GAAG,SAAS,KAAK;CAC7B;CAEA,MAAM,eAAuD,CAAC;CAC9D,MAAM,QAAkB,CAAC;CACzB,MAAM,QAAQ,gBAAgB,IAAI;CAClC,MAAM,OAAmB;EAAE,YAAY,MAAM;EAAM,UAAU,CAAC;CAAE;CAMhE,MAAM,UAA+D,CAAC;CAEtE,KAAK,MAAM,UAAU,QAAQ,SAAS;EACpC,MAAM,OAAO,KAAK,QAAQ,KAAK,aAAa,OAAO;EACnD,IAAI,WAAW,OAAO;GAEpB,KAAK,SAAS,MAAM,kBAAkB,OAAO,eAAe,IAAI,IAAI,CAAC;GACrE,aAAa,MAAM,KAAK;GACxB,QAAQ,KAAK;IAAE;IAAM,aAAa,mBAAmB,MAAM,IAAI;GAAE,CAAC;EACpE,OAAO,IAAI,WAAW,OAAO;GAC3B,MAAM,OAAO,UAAU,IAAI;GAC3B,KAAK,SAAS,MAAM,kBAAkB,OAAO,eAAe,IAAI,IAAI,CAAC;GACrE,aAAa,MAAM,KAAK;GACxB,QAAQ,KAAK;IAAE;IAAM,aAAa,UAAU,MAAM,QAAQ,IAAI,CAAC;GAAE,CAAC;EACpE,OAAO,IAAI,WAAW,aAAa;GAGjC,MAAM,UAAU,0BAA0B,MAAM,EAC9C,YAAY,wBAAwB,UACtC,CAAC;GACD,KAAK,SAAS,YAAY,kBAAkB,OAAO,eAAe,UAAU,OAAO,CAAC;GACpF,aAAa,YAAY,QAAQ;GACjC,QAAQ,KAAK;IAAE;IAAM,aAAa,UAAU,MAAM,QAAQ,OAAO,CAAC;GAAE,CAAC;EACvE,OAAO;GACL,MAAM,QAAQ,WAAW,MAAM,EAAE,YAAY,wBAAwB,IAAI,CAAC;GAC1E,KAAK,SAAS,MAAM,kBAAkB,OAAO,eAAe,IAAI,KAAK,CAAC;GACtE,aAAa,MAAM,MAAM;GACzB,QAAQ,KAAK;IAAE;IAAM,aAAa,UAAU,MAAM,QAAQ,KAAK,CAAC;GAAE,CAAC;EACrE;CACF;CAEA,iBAAiB,IAAI;CAErB,KAAK,MAAM,EAAE,MAAM,WAAW,SAAS;EACrC,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;EAC9C,MAAM,MAAM;EACZ,MAAM,KAAK,IAAI;CACjB;CAEA,MAAM,aAAa,KAAK,QAAQ,KAAK,mBAAmB;CACxD,MAAM,UAAU,YAAY,GAAG,KAAK,UAAU,QAAQ,MAAM,CAAC,EAAE,GAAG;CAClE,MAAM,KAAK,UAAU;CAErB,MAAM,WAAW,KAAK,QAAQ,KAAK,WAAW;CAC9C,MAAM,UACJ,UACA,iBAAiB;EACf,OAAO;EACP,SAAS,QAAQ;EACjB,kBAAkB,QAAQ;EAC1B,aAAa,OAAO,KAAK,UAAU,SAAS,KAAK,CAAC;EAClD,aAAa,OAAO;EACpB,UAAU,OAAO;EACjB;EACA;CACF,CAAC,CACH;CACA,MAAM,KAAK,QAAQ;CAEnB,OAAO;EAAE;EAAQ;EAAM,MAAM,KAAK;EAAQ,OAAO;EAAQ;EAAc;EAAM;CAAM;AACrF;AAEA,SAAgB,gBAAgB,MAAc,QAA0B;CACtE,OAAO;EAAC;EAAmB;EAAU;EAAM;EAAQ;EAAK;EAAe;CAAS;AAClF;AAEA,SAAS,YAAY,MAAc,QAAsB;CAEvD,IADc,UAAU,SAAS,CAAC,iBAAiB,GAAG,EAAE,OAAO,SAAS,CAChE,CAAC,CAAC,WAAW,GACnB,MAAM,IAAI,MACR,gFACF;CAEF,IAAI,CAAC,QAAQ,IAAI,UACf,MAAM,IAAI,MAAM,gDAAgD;CAElE,MAAM,CAAC,SAAS,GAAG,QAAQ,gBAAgB,MAAM,MAAM;CAEvD,MAAM,MAAM,UAAU,SAAS,MAAM,EAAE,OAAO,UAAU,CAAC;CACzD,IAAI,IAAI,WAAW,GAAG,MAAM,IAAI,MAAM,iCAAiC,OAAO,IAAI,MAAM,GAAG;AAC7F;AAOA,MAAM,wBACJ;AAEF,SAAgB,wBAAwB,MAAuC;CAC7E,MAAM,OAA8B;EAClC,QAAQ,CAAC;EACT,KAAK;EACL,SAAS,CAAC,GAAG,eAAe;EAC5B,kBAAkB;EAClB,MAAM;CACR;CACA,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,MAAM,KAAK;EACjB,IAAI,QAAQ,SACV,KAAK,MAAM,KAAK,EAAE,MAAM;OACnB,IAAI,QAAQ,aAAa;GAC9B,MAAM,OAAO,KAAK,EAAE,MAAM,GAAA,CAAI,MAAM,GAAG,CAAC,CAAC,OAAO,OAAO;GACvD,KAAK,MAAM,UAAU,KACnB,IAAI,CAAC,gBAAgB,SAAS,MAAuB,GACnD,MAAM,IAAI,MACR,mBAAmB,OAAO,sBAAsB,gBAAgB,KAAK,GAAG,GAC1E;GAGJ,KAAK,UAAU;EACjB,OAAO,IAAI,QAAQ,uBACjB,KAAK,mBAAmB;OACnB,IAAI,QAAQ,UACjB,KAAK,OAAO,KAAK,EAAE,MAAM;OACpB,IAAI,IAAI,WAAW,IAAI,GAC5B,MAAM,IAAI,MAAM,iBAAiB,IAAI,EAAE;OAEvC,KAAK,OAAO,KAAK,GAAG;CAExB;CACA,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC,KAAK,KACpC,MAAM,IAAI,MAAM,qBAAqB;CAEvC,IAAI,KAAK,SAAS,QAAQ,CAAC,qBAAqB,KAAK,KAAK,IAAI,GAC5D,MAAM,IAAI,MAAM,mCAAmC,KAAK,KAAK,EAAE;CAEjE,OAAO;AACT;;AAGA,eAAsB,qBAAqB,MAAiC;CAC1E,IAAI;CACJ,IAAI;EACF,OAAO,wBAAwB,IAAI;CACrC,SAAS,OAAO;EACd,QAAQ,OAAO,MAAM,GAAG,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,EAAE,GAAG;EAClF,OAAO;CACT;CACA,MAAM,UAAU,MAAM,eAAe,KAAK,QAAQ,IAAI;CACtD,QAAQ,OAAO,MACb,GAAG,KAAK,UACN;EACE,MAAM,QAAQ;EACd,MAAM,QAAQ;EACd,cAAc,QAAQ;EACtB,MAAM,QAAQ;EACd,OAAO,QAAQ;CACjB,GACA,MACA,CACF,EAAE,GACJ;CACA,QAAQ,OAAO,MAAM,aAAa,KAAK,IAAI,IAAI,QAAQ,MAAM,OAAO,UAAU;CAC9E,IAAI,KAAK,SAAS,MAAM;EACtB,YAAY,KAAK,MAAM,KAAK,GAAG;EAC/B,QAAQ,OAAO,MAAM,YAAY,KAAK,KAAK,GAAG;CAChD;CACA,OAAO;AACT"}
|
package/dist/hosted/index.d.ts
CHANGED
|
@@ -1,25 +1,14 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { o as
|
|
3
|
-
import { _ as UnixNanoTimestamp, a as hostedTenantFromEnv, c as EvalRunGenerationSnapshot, d as HostedIngestHeaders, f as HostedWireVersion, g as TraceSpanEvent, h as IngestTracesRequest, i as hostedClientFromEnv, l as EvalRunStatus, m as IngestResponse, n as HostedTenant, o as EvalRunCellScore, p as IngestEvalRunsRequest, r as createHostedClient, s as EvalRunEvent, t as HostedClient, u as HOSTED_WIRE_VERSION } from "../client-kPQYT_56.js";
|
|
1
|
+
import { o as InsightReport } from "../insight-report-DRe8LB6d.js";
|
|
2
|
+
import { _ as UnixNanoTimestamp, a as hostedTenantFromEnv, c as EvalRunGenerationSnapshot, d as HostedIngestHeaders, f as HostedWireVersion, g as TraceSpanEvent, h as IngestTracesRequest, i as hostedClientFromEnv, l as EvalRunStatus, m as IngestResponse, n as HostedTenant, o as EvalRunCellScore, p as IngestEvalRunsRequest, r as createHostedClient, s as EvalRunEvent, t as HostedClient, u as HOSTED_WIRE_VERSION } from "../client-L9VVPkim.js";
|
|
4
3
|
import { z } from "zod";
|
|
5
4
|
//#region src/hosted/schemas.d.ts
|
|
6
5
|
declare const UnixNanoTimestampSchema: z.ZodType<UnixNanoTimestamp>;
|
|
7
6
|
declare const InsightReportSchema: z.ZodType<InsightReport>;
|
|
8
|
-
declare const MutableSurfaceSchema: z.ZodType<MutableSurface>;
|
|
9
|
-
declare const RunTerminalOutcomeSchema: z.ZodEnum<{
|
|
10
|
-
cancelled: "cancelled";
|
|
11
|
-
failed: "failed";
|
|
12
|
-
incomplete: "incomplete";
|
|
13
|
-
succeeded: "succeeded";
|
|
14
|
-
unknown: "unknown";
|
|
15
|
-
}>;
|
|
16
|
-
declare const EvalRunCellScoreSchema: z.ZodType<EvalRunCellScore>;
|
|
17
|
-
declare const EvalRunGenerationSnapshotSchema: z.ZodType<EvalRunGenerationSnapshot>;
|
|
18
7
|
declare const EvalRunEventSchema: z.ZodType<EvalRunEvent>;
|
|
19
8
|
declare const TraceSpanEventSchema: z.ZodType<TraceSpanEvent>;
|
|
20
9
|
declare const IngestEvalRunsRequestSchema: z.ZodType<IngestEvalRunsRequest>;
|
|
21
10
|
declare const IngestTracesRequestSchema: z.ZodType<IngestTracesRequest>;
|
|
22
11
|
declare const IngestResponseSchema: z.ZodType<IngestResponse>;
|
|
23
12
|
//#endregion
|
|
24
|
-
export { type EvalRunCellScore,
|
|
13
|
+
export { type EvalRunCellScore, type EvalRunEvent, EvalRunEventSchema, type EvalRunGenerationSnapshot, type EvalRunStatus, HOSTED_WIRE_VERSION, type HostedClient, type HostedIngestHeaders, type HostedTenant, type HostedWireVersion, type IngestEvalRunsRequest, IngestEvalRunsRequestSchema, type IngestResponse, IngestResponseSchema, type IngestTracesRequest, IngestTracesRequestSchema, type InsightReport, InsightReportSchema, type TraceSpanEvent, TraceSpanEventSchema, type UnixNanoTimestamp, UnixNanoTimestampSchema, createHostedClient, hostedClientFromEnv, hostedTenantFromEnv };
|
|
25
14
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/hosted/schemas.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/hosted/schemas.ts"],"mappings":";;;;cA+Ba,yBAAyB,EAAE,QAAQ;cA6XnC,qBAAqB,EAAE,QAAQ;cA8H/B,oBAAoB,EAAE,QAAQ;cAwD9B,sBAAsB,EAAE,QAAQ;cA0ChC,6BAA6B,EAAE,QAAQ;cAOvC,2BAA2B,EAAE,QAAQ;cAOrC,sBAAsB,EAAE,QAAQ"}
|
package/dist/hosted/index.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import { a as
|
|
2
|
-
export {
|
|
1
|
+
import { a as IngestEvalRunsRequestSchema, c as InsightReportSchema, d as HOSTED_WIRE_VERSION, i as EvalRunEventSchema, l as TraceSpanEventSchema, n as hostedClientFromEnv, o as IngestResponseSchema, r as hostedTenantFromEnv, s as IngestTracesRequestSchema, t as createHostedClient, u as UnixNanoTimestampSchema } from "../client-CX7KqIdB.js";
|
|
2
|
+
export { EvalRunEventSchema, HOSTED_WIRE_VERSION, IngestEvalRunsRequestSchema, IngestResponseSchema, IngestTracesRequestSchema, InsightReportSchema, TraceSpanEventSchema, UnixNanoTimestampSchema, createHostedClient, hostedClientFromEnv, hostedTenantFromEnv };
|
|
@@ -1,13 +1,12 @@
|
|
|
1
1
|
import { o as Severity } from "./multi-layer-verifier-BUaQ4C17.js";
|
|
2
2
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
|
|
3
|
-
import {
|
|
4
|
-
import { i as AnalystFinding, n as AnalystContext } from "./types-
|
|
5
|
-
import { i as TraceAnalystDefinition } from "./default-registry-
|
|
6
|
-
import { l as ExactCapableAnalyst } from "./exact-types-
|
|
7
|
-
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall, r as ExternalOptimizerCallbackLimits } from "./external-optimizer-contracts-
|
|
8
|
-
import { S as SupervisorRunTree, x as SupervisorRunSources } from "./types-
|
|
9
|
-
import {
|
|
10
|
-
import { z } from "zod";
|
|
3
|
+
import { p as ChatClient } from "./types-Bfk0uxRj.js";
|
|
4
|
+
import { i as AnalystFinding, n as AnalystContext } from "./types-D9ssmxKL.js";
|
|
5
|
+
import { i as TraceAnalystDefinition } from "./default-registry-G9CKMNkc.js";
|
|
6
|
+
import { l as ExactCapableAnalyst } from "./exact-types-qnexxJ1Z.js";
|
|
7
|
+
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall, r as ExternalOptimizerCallbackLimits } from "./external-optimizer-contracts-szBJ_1vh.js";
|
|
8
|
+
import { S as SupervisorRunTree, x as SupervisorRunSources } from "./types-DeIUdzNd.js";
|
|
9
|
+
import { t as TraceAnalysisEngine } from "./engine-Cu5qD5Fc.js";
|
|
11
10
|
//#region src/run-score.d.ts
|
|
12
11
|
interface RunScore {
|
|
13
12
|
success: number;
|
|
@@ -130,8 +129,10 @@ interface SemanticConceptJudgeOptions {
|
|
|
130
129
|
maxPerFileChars?: number;
|
|
131
130
|
/** HTML cap. Default 30000. */
|
|
132
131
|
maxHtmlChars?: number;
|
|
133
|
-
/**
|
|
134
|
-
|
|
132
|
+
/** Caller-owned transport. Required: agent-eval executes no paid model. */
|
|
133
|
+
chat: ChatClient;
|
|
134
|
+
/** Endpoint rates used when the transport reports no billed amount. */
|
|
135
|
+
pricing?: CustomTokenPricing;
|
|
135
136
|
costLedger?: CostLedgerHandle;
|
|
136
137
|
costPhase?: string;
|
|
137
138
|
costTags?: Record<string, string>;
|
|
@@ -151,7 +152,7 @@ declare const SEMANTIC_CONCEPT_JUDGE_VERSION = "semantic-concept-judge-v1-2026-0
|
|
|
151
152
|
* LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat
|
|
152
153
|
* that as "skip" rather than "fail."
|
|
153
154
|
*/
|
|
154
|
-
declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, options
|
|
155
|
+
declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, options: SemanticConceptJudgeOptions): Promise<SemanticConceptJudgeResult>;
|
|
155
156
|
//#endregion
|
|
156
157
|
//#region src/analyst/trace-tool-callback.d.ts
|
|
157
158
|
type TraceToolCallbackLimits = ExternalOptimizerCallbackLimits;
|
|
@@ -208,6 +209,40 @@ interface DspyRlmTraceEngineOptions {
|
|
|
208
209
|
declare function createDspyRlmTraceEngine(options: DspyRlmTraceEngineOptions): TraceAnalysisEngine;
|
|
209
210
|
//#endregion
|
|
210
211
|
//#region src/analyst/finding-subject.d.ts
|
|
212
|
+
/**
|
|
213
|
+
* Typed `FindingSubject` — the canonical grammar every analyst kind emits.
|
|
214
|
+
*
|
|
215
|
+
* Background: kind actor prompts have always documented a subject grammar
|
|
216
|
+
* (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
|
|
217
|
+
* LLM was unconstrained — it could emit `subject: "fix the prompt"`
|
|
218
|
+
* (prose) and downstream adapters routed on `startsWith(...)` would
|
|
219
|
+
* silently skip it. Every per-vertical `ImprovementAdapter` had a
|
|
220
|
+
* routing table that mostly caught nothing.
|
|
221
|
+
*
|
|
222
|
+
* This module fixes that:
|
|
223
|
+
* - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
|
|
224
|
+
* when `raw` matches the grammar, else `null`. Used at the
|
|
225
|
+
* `RawAnalystFindingSchema` boundary so malformed subjects are
|
|
226
|
+
* rejected loudly instead of silently lifted into the registry.
|
|
227
|
+
* - `FindingSubjectKind` — the union of valid locus categories. Each
|
|
228
|
+
* variant carries the typed components downstream adapters resolve
|
|
229
|
+
* against the agent's surface manifest (no string parsing in the
|
|
230
|
+
* adapter).
|
|
231
|
+
* - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
|
|
232
|
+
* grammar string embedded in kind actor prompts. Drift between
|
|
233
|
+
* prompt and parser is impossible if every kind imports this.
|
|
234
|
+
*
|
|
235
|
+
* The grammar is intentionally NARROW — only loci the substrate's
|
|
236
|
+
* default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
|
|
237
|
+
* finding with a subject outside this set fails the parser; the kind
|
|
238
|
+
* author either extends the grammar here (and adds adapter routing)
|
|
239
|
+
* or rephrases the prompt to map onto an existing variant.
|
|
240
|
+
*
|
|
241
|
+
* `failure-mode` is the one exception — its subjects are free-form
|
|
242
|
+
* cluster labels, not loci. The schema preserves them as
|
|
243
|
+
* `{ kind: 'cluster', label }` and the adapters skip them (cluster
|
|
244
|
+
* findings are evidence, not actionable mutations).
|
|
245
|
+
*/
|
|
211
246
|
/**
|
|
212
247
|
* Discriminated union of every locus the substrate can route findings to.
|
|
213
248
|
*
|
|
@@ -322,7 +357,6 @@ declare function renderFindingSubject(s: FindingSubject): string;
|
|
|
322
357
|
* lock the table to the parser.
|
|
323
358
|
*/
|
|
324
359
|
declare const FINDING_SUBJECT_SYNTAX: Readonly<Record<FindingSubjectKind, string>>;
|
|
325
|
-
declare const FINDING_SUBJECT_GRAMMAR_PROMPT: string;
|
|
326
360
|
/**
|
|
327
361
|
* The variants each kind is allowed to emit. Used at the kind factory
|
|
328
362
|
* boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
|
|
@@ -334,18 +368,6 @@ declare const FINDING_SUBJECT_GRAMMAR_PROMPT: string;
|
|
|
334
368
|
declare const KIND_EXPECTED_SUBJECTS: Record<string, ReadonlyArray<FindingSubjectKind>>;
|
|
335
369
|
/** Render only the subject forms one analyst kind is permitted to emit. */
|
|
336
370
|
declare function findingSubjectGrammarPromptFor(kindId: string): string;
|
|
337
|
-
/**
|
|
338
|
-
* Zod schema that validates a raw subject string and returns the parsed
|
|
339
|
-
* `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
|
|
340
|
-
* `transform`, so `subject` arrives at the kind factory either as a
|
|
341
|
-
* typed locus or as a parse error attached to a single Zod issue.
|
|
342
|
-
*
|
|
343
|
-
* Optionality is preserved: subjects ARE optional on the wire (some
|
|
344
|
-
* findings are descriptive, not actionable). When present, they MUST
|
|
345
|
-
* parse — emitting a malformed subject is a contract violation, not a
|
|
346
|
-
* soft signal.
|
|
347
|
-
*/
|
|
348
|
-
declare const FindingSubjectStringSchema: z.ZodString;
|
|
349
371
|
//#endregion
|
|
350
372
|
//#region src/analyst/findings-store.d.ts
|
|
351
373
|
/**
|
|
@@ -448,5 +470,5 @@ declare const KNOWLEDGE_POISONING_KIND_SPEC: TraceAnalystDefinition;
|
|
|
448
470
|
*/
|
|
449
471
|
declare const DEFAULT_TRACE_ANALYST_KINDS: readonly TraceAnalystDefinition[];
|
|
450
472
|
//#endregion
|
|
451
|
-
export {
|
|
452
|
-
//# sourceMappingURL=index-
|
|
473
|
+
export { SemanticConceptJudgeResult as A, DspyRlmTraceEngineOptions as C, SEMANTIC_CONCEPT_JUDGE_VERSION as D, ConceptSpec as E, clamp01 as F, RunScore as M, RunScoreWeights as N, SemanticConceptJudgeInput as O, aggregateRunScore as P, renderFindingSubject as S, ConceptFinding as T, FindingSubject as _, FAILURE_MODE_KIND_SPEC as a, findingSubjectGrammarPromptFor as b, emitControlIntegrityFindings as c, FindingsStore as d, PersistedFinding as f, FINDING_SUBJECT_SYNTAX as g, FINDING_SUBJECT_KINDS as h, IMPROVEMENT_KIND_SPEC as i, runSemanticConceptJudge as j, SemanticConceptJudgeOptions as k, DiffPolicy as l, diffFindings as m, KNOWLEDGE_POISONING_KIND_SPEC as n, CONTROL_INTEGRITY_ANALYST as o, defaultIsMaterial as p, KNOWLEDGE_GAP_KIND_SPEC as r, ControlIntegrityAnalyst as s, DEFAULT_TRACE_ANALYST_KINDS as t, FindingsDiff as u, FindingSubjectKind as v, createDspyRlmTraceEngine as w, parseFindingSubject as x, KIND_EXPECTED_SUBJECTS as y };
|
|
474
|
+
//# sourceMappingURL=index-CGtH1piv.d.ts.map
|