@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,763 @@
|
|
|
1
|
+
import { _ as GATE_POLICIES, b as undeclaredStepPayload, g as GATE_CHECK_IDS, i as ROLLOUT_SCHEMA, p as isTrainableSplit, r as ROLLOUT_ROLES, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
|
|
2
|
+
import { l as toVerifiersRolloutOutputs, o as toRftItems, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
|
|
3
|
+
import { basename, dirname, join } from "node:path";
|
|
4
|
+
import { appendFile, mkdir, readFile, writeFile } from "node:fs/promises";
|
|
5
|
+
import { spawnSync } from "node:child_process";
|
|
6
|
+
//#region src/rollout/ledger.ts
|
|
7
|
+
/**
|
|
8
|
+
* Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`
|
|
9
|
+
* lines. Writes validate BEFORE touching disk (a bad line never lands);
|
|
10
|
+
* reads validate line-by-line and fail loud with the line number, because a
|
|
11
|
+
* silently-skipped rollout is a corrupted dataset.
|
|
12
|
+
*
|
|
13
|
+
* "Validate" includes the anti-Goodhart invariant (a realness-gated line may
|
|
14
|
+
* not carry a positive reward), so a poisoned line can neither enter a ledger
|
|
15
|
+
* nor leave one.
|
|
16
|
+
*
|
|
17
|
+
* Two read modes, matching the two write-side row classes: `readRolloutLedger`
|
|
18
|
+
* re-validates under the mint policy (training data), `readRolloutJournal`
|
|
19
|
+
* under the write policy (supervision journals, whose unscreened positive
|
|
20
|
+
* rewards are writable and must stay readable).
|
|
21
|
+
*/
|
|
22
|
+
function serialize(lines) {
|
|
23
|
+
for (const [i, line] of lines.entries()) assertRolloutLine(line, `rollout line [${i}]`);
|
|
24
|
+
return lines.map((line) => JSON.stringify(line)).join("\n") + (lines.length > 0 ? "\n" : "");
|
|
25
|
+
}
|
|
26
|
+
/** Replace the ledger file with exactly `lines`. */
|
|
27
|
+
async function writeRolloutLedger(path, lines) {
|
|
28
|
+
const payload = serialize(lines);
|
|
29
|
+
await mkdir(dirname(path), { recursive: true });
|
|
30
|
+
await writeFile(path, payload);
|
|
31
|
+
}
|
|
32
|
+
/** Append `lines` to the ledger file (created if absent). */
|
|
33
|
+
async function appendRolloutLines(path, lines) {
|
|
34
|
+
if (lines.length === 0) return;
|
|
35
|
+
const payload = serialize(lines);
|
|
36
|
+
await mkdir(dirname(path), { recursive: true });
|
|
37
|
+
await appendFile(path, payload);
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Read and validate every line. Throws on the first malformed/invalid line
|
|
41
|
+
* (with its 1-based line number) — fail-closed, never a silent drop.
|
|
42
|
+
*
|
|
43
|
+
* Validation includes the anti-Goodhart invariant, which is why the result is
|
|
44
|
+
* `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this
|
|
45
|
+
* process from outside the type system (another run, another machine, a
|
|
46
|
+
* hand-edited JSONL), so this read is the runtime boundary where a poisoned
|
|
47
|
+
* line is refused rather than exported.
|
|
48
|
+
*/
|
|
49
|
+
async function readRolloutLedger(path) {
|
|
50
|
+
return readLines(path, (parsed, context) => assertMinted(parsed, context));
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Read a ledger under the WRITE-side policy (`validateRolloutLine`), which
|
|
54
|
+
* omits the unscreened-reward check. `writeRolloutLedger` accepts a
|
|
55
|
+
* supervision-journal row (`realness_screened: false` with a positive reward
|
|
56
|
+
* — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says
|
|
57
|
+
* such rows "must stay writable, readable and reportable"; a read API that
|
|
58
|
+
* only re-validated under `assertMinted` made every such file unreadable —
|
|
59
|
+
* write-accepted but read-refused is a data-loss trap.
|
|
60
|
+
*
|
|
61
|
+
* The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here
|
|
62
|
+
* can reach a training exporter without passing `assertMinted`, so the
|
|
63
|
+
* promotion gate (which DOES enforce unscreened-reward) is exactly as closed
|
|
64
|
+
* as before. Use `readRolloutLedger` when the file is training data.
|
|
65
|
+
*/
|
|
66
|
+
async function readRolloutJournal(path) {
|
|
67
|
+
return readLines(path, (parsed, context) => {
|
|
68
|
+
assertRolloutLine(parsed, context);
|
|
69
|
+
return parsed;
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
async function readLines(path, admit) {
|
|
73
|
+
const raw = await readFile(path, "utf8");
|
|
74
|
+
const lines = [];
|
|
75
|
+
const rawLines = raw.split("\n");
|
|
76
|
+
for (let i = 0; i < rawLines.length; i++) {
|
|
77
|
+
const text = rawLines[i];
|
|
78
|
+
if (!text?.trim()) continue;
|
|
79
|
+
let parsed;
|
|
80
|
+
try {
|
|
81
|
+
parsed = JSON.parse(text);
|
|
82
|
+
} catch (error) {
|
|
83
|
+
throw new Error(`${path}:${i + 1}: malformed JSON — ${error instanceof Error ? error.message : String(error)}`);
|
|
84
|
+
}
|
|
85
|
+
lines.push(admit(parsed, `${path}:${i + 1}`));
|
|
86
|
+
}
|
|
87
|
+
return lines;
|
|
88
|
+
}
|
|
89
|
+
//#endregion
|
|
90
|
+
//#region src/rollout/release/gate-report.ts
|
|
91
|
+
const FORMAT_GATE_DISPOSITION = {
|
|
92
|
+
sft: "exclude",
|
|
93
|
+
verifiers: "zero-and-flag",
|
|
94
|
+
rft: "zero-and-flag",
|
|
95
|
+
raw: "zero-and-flag"
|
|
96
|
+
};
|
|
97
|
+
/** Rollout ids of every gated line, the key the emitted rows are matched on. */
|
|
98
|
+
function gatedRolloutIds(lines) {
|
|
99
|
+
return new Set(lines.filter((line) => line.outcome.realness_gated).map((line) => line.rollout_id));
|
|
100
|
+
}
|
|
101
|
+
/**
|
|
102
|
+
* Row refs per format. Written as one adapter per format so that the knowledge
|
|
103
|
+
* of WHERE the id and reward live in each published shape sits next to the
|
|
104
|
+
* assertion that uses it — an exporter that moves either field breaks here
|
|
105
|
+
* rather than silently reporting zero gated rows.
|
|
106
|
+
*/
|
|
107
|
+
const releaseRowRefs = {
|
|
108
|
+
sft: (rows) => rows.map((row) => ({
|
|
109
|
+
rollout_id: row.metadata.rollout_id,
|
|
110
|
+
reward: row.metadata.reward,
|
|
111
|
+
realness_screened: row.metadata.realness_screened
|
|
112
|
+
})),
|
|
113
|
+
verifiers: (rows) => rows.map((row) => ({
|
|
114
|
+
rollout_id: row.info.rollout_id,
|
|
115
|
+
reward: row.reward,
|
|
116
|
+
evidence: row.metrics,
|
|
117
|
+
realness_screened: row.info.realness_screened
|
|
118
|
+
})),
|
|
119
|
+
rft: (rows) => rows.map((row) => ({
|
|
120
|
+
rollout_id: row.reference.rollout_id,
|
|
121
|
+
reward: row.reference.reward,
|
|
122
|
+
evidence: row.reference.verdict,
|
|
123
|
+
realness_screened: row.reference.realness_screened
|
|
124
|
+
})),
|
|
125
|
+
raw: (lines) => lines.map((line) => ({
|
|
126
|
+
rollout_id: line.rollout_id,
|
|
127
|
+
reward: line.outcome.reward,
|
|
128
|
+
evidence: {
|
|
129
|
+
metrics: line.outcome.metrics,
|
|
130
|
+
verdict: line.outcome.verdict
|
|
131
|
+
},
|
|
132
|
+
stepEvidence: undeclaredStepPayload(line.steps),
|
|
133
|
+
realness_screened: line.outcome.realness_screened ?? null
|
|
134
|
+
}))
|
|
135
|
+
};
|
|
136
|
+
/** Every positive finite number inside a row's reward-derived payload, with its path. */
|
|
137
|
+
function positiveNumbersIn(value, path) {
|
|
138
|
+
if (typeof value === "number") return Number.isFinite(value) && value > 0 ? [{
|
|
139
|
+
path,
|
|
140
|
+
value
|
|
141
|
+
}] : [];
|
|
142
|
+
if (Array.isArray(value)) return value.flatMap((item, i) => positiveNumbersIn(item, `${path}[${i}]`));
|
|
143
|
+
if (typeof value === "object" && value !== null) return Object.entries(value).flatMap(([key, child]) => positiveNumbersIn(child, path === "" ? key : `${path}.${key}`));
|
|
144
|
+
return [];
|
|
145
|
+
}
|
|
146
|
+
/** Measure one format's gated rows from the refs of the rows about to be written. */
|
|
147
|
+
function measureFormatGate(gated, refs) {
|
|
148
|
+
const emitted = refs.filter((ref) => gated.has(ref.rollout_id));
|
|
149
|
+
const rewards = emitted.map((ref) => ref.reward).filter((r) => r !== null);
|
|
150
|
+
const evidence = emitted.flatMap((ref) => positiveNumbersIn(ref.evidence, ""));
|
|
151
|
+
const stepEvidence = emitted.flatMap((ref) => positiveNumbersIn(ref.stepEvidence, "steps"));
|
|
152
|
+
const unscreened = refs.filter((ref) => ref.realness_screened === false).map((ref) => ref.reward).filter((r) => r !== null && r > 0);
|
|
153
|
+
return {
|
|
154
|
+
input: gated.size,
|
|
155
|
+
emitted: emitted.length,
|
|
156
|
+
excluded: gated.size - emitted.length,
|
|
157
|
+
maxEmittedReward: rewards.length === 0 ? null : Math.max(...rewards),
|
|
158
|
+
maxEmittedEvidence: evidence.reduce((best, found) => best === null || found.value > best.value ? found : best, null),
|
|
159
|
+
unscreenedPositiveRows: unscreened.length,
|
|
160
|
+
maxUnscreenedReward: unscreened.length === 0 ? null : Math.max(...unscreened),
|
|
161
|
+
maxEmittedStepEvidence: stepEvidence.reduce((best, found) => best === null || found.value > best.value ? found : best, null)
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* The measured form of each canonical gate check, over the rows a release is
|
|
166
|
+
* ABOUT TO WRITE. Returns the failure message, or `null` when the format is
|
|
167
|
+
* clean on that check.
|
|
168
|
+
*
|
|
169
|
+
* TOTAL over `GateCheckId` — this map and `GATE_POLICIES.assertGateReport` are
|
|
170
|
+
* the two things a new check breaks here, so the release certifier cannot be
|
|
171
|
+
* left behind by a check added anywhere else in the package. That is the whole
|
|
172
|
+
* point: for four rounds each guard composed its own subset by hand, and a
|
|
173
|
+
* release certifying CLEAN while leaking is the most expensive version of that
|
|
174
|
+
* mistake, because it is the leak plus a document saying there isn't one.
|
|
175
|
+
*/
|
|
176
|
+
const REPORT_MEASURES = {
|
|
177
|
+
"reward-relationship": (format, counts) => counts.maxEmittedReward === null || counts.maxEmittedReward <= 0 ? null : `release format "${format}": ${counts.emitted} realness-gated row(s) carry a positive reward (max ${counts.maxEmittedReward}). A run flagged as gamed may not ship a positive reward in any config.`,
|
|
178
|
+
"gated-evidence": (format, counts) => {
|
|
179
|
+
if (counts.maxEmittedEvidence === null) return null;
|
|
180
|
+
const { path, value } = counts.maxEmittedEvidence;
|
|
181
|
+
return `release format "${format}": ${counts.emitted} realness-gated row(s) carry a positive reward-derived number (${path} = ${value}). Zeroing the scalar is not enough — the per-layer scores and judge verdict a fabricated reward was computed FROM are the same signal in component form, and in this format they are read as training input. They belong in \`provenance.gated_evidence\` (see \`gateGamedOutcome\`), not on the row.`;
|
|
182
|
+
},
|
|
183
|
+
"undeclared-step-payload": (format, counts) => {
|
|
184
|
+
if (counts.maxEmittedStepEvidence === null) return null;
|
|
185
|
+
const { path, value } = counts.maxEmittedStepEvidence;
|
|
186
|
+
return `release format "${format}": ${counts.emitted} realness-gated row(s) carry a positive per-step number under a field \`tangle.rollout.v1\` does not declare (${path} = ${value}). A per-step reward is the same training signal as the scalar, in credit-assignment form, and the exporters copy \`steps\` through verbatim — so the row ships it beside a \`reward\` of 0. It belongs in \`provenance.gated_evidence.steps\` (see \`gateGamedOutcome\`).`;
|
|
187
|
+
},
|
|
188
|
+
"unscreened-reward": (format, counts) => counts.maxUnscreenedReward === null ? null : `release format "${format}": ${counts.unscreenedPositiveRows} row(s) carry a positive reward (max ${counts.maxUnscreenedReward}) whose producer declares that NO authenticity screen ran on it (\`realness_screened: false\`). Nothing has established those successes are real, and a published dataset may not present an unqualified verdict as a measured one. Screen the runs, or publish them at \`reward: null\`.`
|
|
189
|
+
};
|
|
190
|
+
/**
|
|
191
|
+
* Fail the build when the measurement disagrees with the declared policy.
|
|
192
|
+
*
|
|
193
|
+
* Throws, never filters: an emitted positive reward on a gated row means an
|
|
194
|
+
* exporter upstream stopped applying the gate, and silently dropping the row
|
|
195
|
+
* would hide the producer that made it — the producer is the actual defect.
|
|
196
|
+
*
|
|
197
|
+
* Certifies the whole emitted outcome, not `reward` alone. The earlier version
|
|
198
|
+
* checked one field and therefore certified a release CLEAN while its
|
|
199
|
+
* `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in
|
|
200
|
+
* the top-level `metrics` dict — the card then rendered "max reward | 0" over
|
|
201
|
+
* exactly that file. A certification that is wrong is worse than an
|
|
202
|
+
* uncertified leak, so the checks it runs are no longer written down here at
|
|
203
|
+
* all: it iterates `GATE_CHECK_IDS` under its own declared policy.
|
|
204
|
+
*/
|
|
205
|
+
function assertGateReport(report) {
|
|
206
|
+
for (const [format, counts] of Object.entries(report.byFormat)) {
|
|
207
|
+
for (const id of GATE_CHECK_IDS) {
|
|
208
|
+
if (GATE_POLICIES.assertGateReport[id].kind !== "enforce") continue;
|
|
209
|
+
const failure = REPORT_MEASURES[id](format, counts);
|
|
210
|
+
if (failure !== null) throw new Error(failure);
|
|
211
|
+
}
|
|
212
|
+
if (FORMAT_GATE_DISPOSITION[format] === "exclude" && counts.emitted > 0) throw new Error(`release format "${format}" declares gated lines EXCLUDED but wrote ${counts.emitted} of them.`);
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
//#endregion
|
|
216
|
+
//#region src/rollout/release/card.ts
|
|
217
|
+
/**
|
|
218
|
+
* HuggingFace dataset-card (README.md) generation for a rollout-ledger release.
|
|
219
|
+
*
|
|
220
|
+
* The card is a pure function of the SCRUBBED lines plus the release options —
|
|
221
|
+
* no timestamps, no environment reads — so rebuilding from the same ledger
|
|
222
|
+
* yields byte-identical output. It documents the schema, provenance (run ids,
|
|
223
|
+
* generations, the official judge), per-role reward semantics including the
|
|
224
|
+
* inherited/contribution caveat, and a role × reward counts table.
|
|
225
|
+
*/
|
|
226
|
+
const RELEASE_FORMATS = [
|
|
227
|
+
"sft",
|
|
228
|
+
"verifiers",
|
|
229
|
+
"rft",
|
|
230
|
+
"raw"
|
|
231
|
+
];
|
|
232
|
+
/** Format → data file path inside the dataset dir (train split only). */
|
|
233
|
+
const FORMAT_FILES = {
|
|
234
|
+
sft: "sft/train.jsonl",
|
|
235
|
+
verifiers: "verifiers/train.jsonl",
|
|
236
|
+
rft: "rft/train.jsonl",
|
|
237
|
+
raw: "raw/train.jsonl"
|
|
238
|
+
};
|
|
239
|
+
const FORMAT_DESCRIPTIONS = {
|
|
240
|
+
sft: "Successful trainable-split transcripts (`reward >= 1`, never realness-gated) as `{messages, metadata}` chat JSONL.",
|
|
241
|
+
verifiers: "Prime Intellect verifiers `RolloutOutput`: prompt/completion split at the first assistant turn, plus reward, metrics, tool defs, and token usage.",
|
|
242
|
+
rft: "OpenAI RFT items: prompt turns plus `reference.*` verdict fields for a grader (completions are re-sampled during RFT).",
|
|
243
|
+
raw: `Full \`${ROLLOUT_SCHEMA}\` ledger lines (scrubbed), one per agent invocation.`
|
|
244
|
+
};
|
|
245
|
+
function unique(values) {
|
|
246
|
+
return [...new Set(values.filter((v) => v !== null))].sort();
|
|
247
|
+
}
|
|
248
|
+
function formatReward(reward) {
|
|
249
|
+
if (reward === null) return "null";
|
|
250
|
+
return Number.isInteger(reward) ? String(reward) : reward.toFixed(4);
|
|
251
|
+
}
|
|
252
|
+
function markdownTable(header, rows) {
|
|
253
|
+
return [
|
|
254
|
+
`| ${header.join(" | ")} |`,
|
|
255
|
+
`| ${header.map(() => "---").join(" | ")} |`,
|
|
256
|
+
...rows.map((row) => `| ${row.join(" | ")} |`)
|
|
257
|
+
].join("\n");
|
|
258
|
+
}
|
|
259
|
+
/** Where the anti-Goodhart flag lives on each format's emitted row. */
|
|
260
|
+
const GATE_FLAG_FIELD = {
|
|
261
|
+
sft: "— (no gated row is written)",
|
|
262
|
+
verifiers: "`info.realness_gated`",
|
|
263
|
+
rft: "`reference.realness_gated`",
|
|
264
|
+
raw: "`outcome.realness_gated`"
|
|
265
|
+
};
|
|
266
|
+
const GATE_POLICY_LABEL = {
|
|
267
|
+
sft: "EXCLUDED — an SFT row is an imitation target",
|
|
268
|
+
verifiers: "INCLUDED, reward forced to 0, flagged",
|
|
269
|
+
rft: "INCLUDED, reward forced to 0, flagged",
|
|
270
|
+
raw: "INCLUDED verbatim, flagged"
|
|
271
|
+
};
|
|
272
|
+
function gateSection(formats, gate, totalLines) {
|
|
273
|
+
const rows = formats.map((format) => {
|
|
274
|
+
const counts = gate.byFormat[format];
|
|
275
|
+
return [
|
|
276
|
+
format,
|
|
277
|
+
GATE_POLICY_LABEL[format],
|
|
278
|
+
String(counts?.emitted ?? 0),
|
|
279
|
+
String(counts?.excluded ?? 0),
|
|
280
|
+
counts?.maxEmittedReward === null || counts === void 0 ? "—" : formatReward(counts.maxEmittedReward),
|
|
281
|
+
counts?.maxEmittedEvidence == null ? "—" : `${counts.maxEmittedEvidence.path} = ${counts.maxEmittedEvidence.value}`,
|
|
282
|
+
GATE_FLAG_FIELD[format]
|
|
283
|
+
];
|
|
284
|
+
});
|
|
285
|
+
const mixedExclusion = formats.some((format) => FORMAT_GATE_DISPOSITION[format] === "zero-and-flag");
|
|
286
|
+
return [
|
|
287
|
+
`\`outcome.realness_gated: true\` marks a run whose success signal was faked (it satisfied the proxy without doing the work). **${gate.gatedLines} of ${totalLines} lines in this release are gated.**`,
|
|
288
|
+
"",
|
|
289
|
+
"The gate is enforced, not asserted. A line pairing `realness_gated: true` with a reward above 0 is rejected by the schema validator, so it cannot be read out of a source ledger or written into `raw/train.jsonl` at all; on a gated line the numbers the reward was computed from (`outcome.metrics`, the verbatim judge `outcome.verdict`, and any per-step field the schema does not declare — a per-step reward is the same signal in credit-assignment form) are moved out of `outcome` and `steps[]` and into `provenance.gated_evidence`, where no config reads them as training input; the exporters re-check the same invariant on every line; and the release build measures the rows it is about to write and refuses to write ANY file if a gated row carries a positive reward — or any positive number derived from one — in ANY config.",
|
|
290
|
+
"",
|
|
291
|
+
"The table below is **measured on the rows in this release**, not a description of intent — the build counts the emitted rows and this card renders those counts:",
|
|
292
|
+
"",
|
|
293
|
+
markdownTable([
|
|
294
|
+
"config",
|
|
295
|
+
"policy",
|
|
296
|
+
"gated rows written",
|
|
297
|
+
"gated rows not written",
|
|
298
|
+
"max reward",
|
|
299
|
+
"max reward-derived number",
|
|
300
|
+
"flag"
|
|
301
|
+
], rows),
|
|
302
|
+
"",
|
|
303
|
+
"Why gated rows are kept where they are kept: an SFT row is imitated verbatim, so a gamed trajectory must never appear in one at any weight. In `verifiers` the reward is a signed learning signal, and a gamed trajectory at reward 0 is a correct negative — dropping it would bias the negative population toward honest failures and leave a trainer no example of gaming being penalized. `rft` re-samples the completion, so only the prompt and the grader reference ship. `raw` is a faithful audit dump, where the gated row is the one an auditor most wants.",
|
|
304
|
+
"",
|
|
305
|
+
"Reward 0 is never the only label: every included config carries the flag on the row itself, because zeroing alone makes a faked success indistinguishable from an honest failure. Filter on the flag to drop the gamed population, or select on it to mine it. The gated run's own measurements are not destroyed either — they are parked verbatim under `provenance.gated_evidence` in `raw/train.jsonl`, which is where an auditor can see what the run claimed and why it was flagged.",
|
|
306
|
+
mixedExclusion ? "\n\"Gated rows not written\" is not all gate: `verifiers` also drops gap lines (empty transcript) and `rft` drops lines with no prompt turn, so that column can mix both causes." : ""
|
|
307
|
+
].join("\n").trimEnd();
|
|
308
|
+
}
|
|
309
|
+
function roleRewardRows(lines) {
|
|
310
|
+
const counts = /* @__PURE__ */ new Map();
|
|
311
|
+
for (const line of lines) {
|
|
312
|
+
const byReward = counts.get(line.role) ?? /* @__PURE__ */ new Map();
|
|
313
|
+
const key = formatReward(line.outcome.reward);
|
|
314
|
+
byReward.set(key, (byReward.get(key) ?? 0) + 1);
|
|
315
|
+
counts.set(line.role, byReward);
|
|
316
|
+
}
|
|
317
|
+
const rows = [];
|
|
318
|
+
for (const role of ROLLOUT_ROLES) {
|
|
319
|
+
const byReward = counts.get(role);
|
|
320
|
+
if (!byReward) continue;
|
|
321
|
+
const keys = [...byReward.keys()].sort((a, b) => {
|
|
322
|
+
if (a === "null") return 1;
|
|
323
|
+
if (b === "null") return -1;
|
|
324
|
+
return Number(b) - Number(a);
|
|
325
|
+
});
|
|
326
|
+
for (const key of keys) rows.push([
|
|
327
|
+
role,
|
|
328
|
+
key,
|
|
329
|
+
String(byReward.get(key))
|
|
330
|
+
]);
|
|
331
|
+
}
|
|
332
|
+
return rows;
|
|
333
|
+
}
|
|
334
|
+
function buildDatasetCard(inputs) {
|
|
335
|
+
const { lines, formats, includeProposers, sourceFiles, scrubTotals, excluded, formatCounts, gate } = inputs;
|
|
336
|
+
assertGateReport(gate);
|
|
337
|
+
const runIds = unique(lines.map((line) => line.run_id));
|
|
338
|
+
const generations = [...new Set(lines.map((line) => line.generation))].filter((g) => g !== null).sort((a, b) => a - b);
|
|
339
|
+
const models = unique(lines.map((line) => line.policy.model));
|
|
340
|
+
const harnesses = unique(lines.map((line) => line.policy.harness));
|
|
341
|
+
const captures = unique(lines.map((line) => line.provenance.capture));
|
|
342
|
+
const rewardSources = unique(lines.map((line) => line.outcome.reward_source));
|
|
343
|
+
const gapLines = lines.filter((line) => line.messages.length === 0).length;
|
|
344
|
+
const gatedLines = lines.filter((line) => line.outcome.realness_gated).length;
|
|
345
|
+
if (gatedLines !== gate.gatedLines) throw new Error(`dataset card: gate report claims ${gate.gatedLines} realness-gated line(s) but the lines it describes contain ${gatedLines} — the card and the build measured different data.`);
|
|
346
|
+
const frontmatter = [
|
|
347
|
+
"---",
|
|
348
|
+
"license: unknown",
|
|
349
|
+
"pretty_name: Tangle rollout ledger — agent trajectories",
|
|
350
|
+
"configs:",
|
|
351
|
+
formats.map((format) => [
|
|
352
|
+
` - config_name: ${format}`,
|
|
353
|
+
" data_files:",
|
|
354
|
+
" - split: train",
|
|
355
|
+
` path: ${FORMAT_FILES[format]}`
|
|
356
|
+
].join("\n")).join("\n"),
|
|
357
|
+
"---"
|
|
358
|
+
].join("\n");
|
|
359
|
+
const formatsTable = markdownTable([
|
|
360
|
+
"config",
|
|
361
|
+
"path",
|
|
362
|
+
"rows",
|
|
363
|
+
"contents"
|
|
364
|
+
], formats.map((format) => [
|
|
365
|
+
format,
|
|
366
|
+
`\`${FORMAT_FILES[format]}\``,
|
|
367
|
+
String(formatCounts[format] ?? 0),
|
|
368
|
+
FORMAT_DESCRIPTIONS[format]
|
|
369
|
+
]));
|
|
370
|
+
const countsTable = markdownTable([
|
|
371
|
+
"role",
|
|
372
|
+
"reward",
|
|
373
|
+
"lines"
|
|
374
|
+
], roleRewardRows(lines));
|
|
375
|
+
const scrubTable = markdownTable(["rule", "rewrites"], Object.entries(scrubTotals).map(([rule, count]) => [rule, String(count)]));
|
|
376
|
+
const proposerNote = includeProposers ? "Proposer sessions are INCLUDED (`--include-proposers`); their transcripts contain improvement-loop harness source." : `Proposer sessions are excluded by default (${excluded.proposers} lines dropped); they contain improvement-loop harness source. Rebuild with \`--include-proposers\` to keep them.`;
|
|
377
|
+
return `${frontmatter}
|
|
378
|
+
|
|
379
|
+
# Tangle rollout ledger — agent trajectories
|
|
380
|
+
|
|
381
|
+
One line per agent invocation (supervisor episode, worker session, proposer shot, judge call, analyst pass) captured by the \`${ROLLOUT_SCHEMA}\` rollout ledger, labeled with improvement-loop coordinates and the official-judge reward, with the full message transcript inline.
|
|
382
|
+
|
|
383
|
+
This release contains the **trainable split only** (\`search\`). Holdout, dev, and canary splits are structurally excluded at build time, and the build additionally drops any non-trainable line as a fail-closed filter (${excluded.nonTrain} dropped here).
|
|
384
|
+
|
|
385
|
+
## Formats
|
|
386
|
+
|
|
387
|
+
${formatsTable}
|
|
388
|
+
|
|
389
|
+
## Anti-Goodhart gate (\`realness_gated\`)
|
|
390
|
+
|
|
391
|
+
${gateSection(formats, gate, lines.length)}
|
|
392
|
+
|
|
393
|
+
## Schema (\`${ROLLOUT_SCHEMA}\`)
|
|
394
|
+
|
|
395
|
+
Each raw line carries:
|
|
396
|
+
|
|
397
|
+
- \`rollout_id\` / \`parent_rollout_id\` — invocation identity; workers point at their spawning supervisor episode.
|
|
398
|
+
- \`run_id\`, \`experiment_id\`, \`candidate_id\` — run/experiment/candidate identity from the producing RunRecord, when present.
|
|
399
|
+
- \`generation\`, \`candidate_index\` — improvement-loop coordinates (\`-1\` = baseline campaign; \`null\` = not an improvement loop).
|
|
400
|
+
- \`role\` — one of ${ROLLOUT_ROLES.map((role) => `\`${role}\``).join(", ")}.
|
|
401
|
+
- \`task\` — suite, instance id, split, seed, replicate index.
|
|
402
|
+
- \`policy\` — harness, model, provider, profile commit, prompt/config hashes, sampling params.
|
|
403
|
+
- \`messages\` / \`tool_defs\` — full transcript in canonical OpenAI chat-with-tools form (including \`reasoning_content\`). An empty \`messages\` array is a labeled gap line; \`provenance.gap\` says why the transcript could not be recovered (${gapLines} gap lines in this release).
|
|
404
|
+
- \`outcome\` — \`reward\` (the single scalar), \`reward_source\`, the verbatim judge \`verdict\`, non-scalar \`metrics\`, and \`realness_gated\` (anti-Goodhart flag; see [Anti-Goodhart gate](#anti-goodhart-gate-realness_gated) for the per-config treatment and the measured counts).
|
|
405
|
+
- \`cost\` — usd, token counts, wall time.
|
|
406
|
+
- \`artifacts\` / \`provenance\` — patch/run-dir/transcript pointers (scrubbed) and capture metadata.
|
|
407
|
+
|
|
408
|
+
## Provenance
|
|
409
|
+
|
|
410
|
+
- Source ledgers: ${sourceFiles.map((file) => `\`${file}\``).join(", ")}
|
|
411
|
+
- Run ids: ${runIds.map((id) => `\`${id}\``).join(", ")}
|
|
412
|
+
- Generations: ${generations.join(", ")} (\`-1\` = baseline campaign)
|
|
413
|
+
- Models: ${models.map((m) => `\`${m}\``).join(", ")}
|
|
414
|
+
- Harnesses: ${harnesses.map((h) => `\`${h}\``).join(", ")}
|
|
415
|
+
- Capture modes: ${captures.join(", ")}
|
|
416
|
+
- Every reward traces to a named source (reward sources in this release: ${rewardSources.map((s) => `\`${s}\``).join(", ")}).
|
|
417
|
+
|
|
418
|
+
${proposerNote}
|
|
419
|
+
|
|
420
|
+
## Reward semantics per role
|
|
421
|
+
|
|
422
|
+
- **agent** — the producing RunRecord's holdout/search score, with the realness gate forcing gamed successes to 0.
|
|
423
|
+
- **supervisor** — the official-judge verdict on the episode's delivered artifact (1 = resolved, 0 = not).
|
|
424
|
+
- **worker** — INHERITED from the parent supervisor episode (\`…/inherited\`). Caveat: reward 1 does not establish this worker's individual contribution (sibling workers in the same episode share the episode outcome), and reward 0 does not prove this worker failed.
|
|
425
|
+
- **proposer** — the fraction of improvement-set instances the proposed candidate resolved (\`…/candidate-resolved-fraction\`); a scalar in [0, 1], not a binary verdict.
|
|
426
|
+
- **judge / analyst** — carry the episode verdict where one applies; otherwise \`reward: null\` (a labeled gap, never 0).
|
|
427
|
+
|
|
428
|
+
## Counts
|
|
429
|
+
|
|
430
|
+
${countsTable}
|
|
431
|
+
|
|
432
|
+
Total lines: ${lines.length}
|
|
433
|
+
|
|
434
|
+
## Scrubbing
|
|
435
|
+
|
|
436
|
+
Absolute home paths were rewritten to \`$WORK\`, credential-shaped strings were replaced with \`[REDACTED:<kind>]\` markers, internal hostnames were normalized to \`*.internal.example\`, and username-bearing incidentals (\`ls -l\` owner columns, per-user pytest tmpdirs) were normalized to \`user\` / \`$USER\`. Rewrite counts for this release (full per-file breakdown in \`scrub-report.json\`):
|
|
437
|
+
|
|
438
|
+
${scrubTable}
|
|
439
|
+
|
|
440
|
+
## License
|
|
441
|
+
|
|
442
|
+
\`license: unknown\` is a placeholder — the releasing operator must set the real SPDX license id in the frontmatter above before publishing.
|
|
443
|
+
|
|
444
|
+
## Citation
|
|
445
|
+
|
|
446
|
+
\`\`\`bibtex
|
|
447
|
+
@misc{tangle_rollout_ledger,
|
|
448
|
+
title = {Tangle rollout ledger — agent trajectories},
|
|
449
|
+
author = {{Tangle Network}},
|
|
450
|
+
howpublished = {HuggingFace Datasets},
|
|
451
|
+
note = {Operator: fill in the repository URL, authors, and year before publishing}
|
|
452
|
+
}
|
|
453
|
+
\`\`\`
|
|
454
|
+
`;
|
|
455
|
+
}
|
|
456
|
+
//#endregion
|
|
457
|
+
//#region src/rollout/release/scrub.ts
|
|
458
|
+
/**
|
|
459
|
+
* Deterministic scrubbing pass over rollout-ledger lines before public release.
|
|
460
|
+
*
|
|
461
|
+
* Every rule is a pure regex rewrite applied to every string value in a line
|
|
462
|
+
* (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed
|
|
463
|
+
* line is still a valid `tangle.rollout.v1` line. Rules are idempotent:
|
|
464
|
+
* scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is
|
|
465
|
+
* the property the release pipeline relies on to prove nothing half-scrubbed
|
|
466
|
+
* ships. Rule order matters: whole `KEY=value` env pairs are redacted before
|
|
467
|
+
* the bare-key rule so one secret is never counted twice.
|
|
468
|
+
*/
|
|
469
|
+
const SCRUB_RULES = [
|
|
470
|
+
{
|
|
471
|
+
name: "home-path",
|
|
472
|
+
pattern: /\/(?:home|Users)\/[A-Za-z0-9._-]+/g,
|
|
473
|
+
rewrite: () => "$WORK"
|
|
474
|
+
},
|
|
475
|
+
{
|
|
476
|
+
name: "home-path-encoded",
|
|
477
|
+
pattern: /(?<=\/)-(?:home|Users)-[A-Za-z0-9_.]+/g,
|
|
478
|
+
rewrite: () => "$WORK"
|
|
479
|
+
},
|
|
480
|
+
{
|
|
481
|
+
name: "tmp-user-dir",
|
|
482
|
+
pattern: /\/tmp\/pytest-of-[A-Za-z0-9._-]+/g,
|
|
483
|
+
rewrite: () => "/tmp/pytest-of-$USER"
|
|
484
|
+
},
|
|
485
|
+
{
|
|
486
|
+
name: "ls-owner",
|
|
487
|
+
pattern: /([-bcdlps][-rwxsStT]{9}[.+@]?\s+\d+\s+)(?!user user(?=\s))[A-Za-z0-9._-]+\s+[A-Za-z0-9._-]+(?=\s)/g,
|
|
488
|
+
rewrite: (_match, prefix) => `${prefix}user user`
|
|
489
|
+
},
|
|
490
|
+
{
|
|
491
|
+
name: "env-secret",
|
|
492
|
+
pattern: /\b([A-Z][A-Z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIALS?))=(?!\[REDACTED:)("[^"]*"|'[^']*'|[^\s"']+)/g,
|
|
493
|
+
rewrite: (_match, name) => `${name}=[REDACTED:env]`
|
|
494
|
+
},
|
|
495
|
+
{
|
|
496
|
+
name: "bearer-token",
|
|
497
|
+
pattern: /\b(Bearer|Basic)\s+(?!\[REDACTED:)[A-Za-z0-9\-._~+/=]{8,}/g,
|
|
498
|
+
rewrite: (_match, scheme) => `${scheme} [REDACTED:bearer]`
|
|
499
|
+
},
|
|
500
|
+
{
|
|
501
|
+
name: "api-key",
|
|
502
|
+
pattern: /\b(?:sk-[A-Za-z0-9_-]{16,}|gh[pousr]_[A-Za-z0-9]{16,}|github_pat_[A-Za-z0-9_]{20,}|hf_[A-Za-z0-9]{16,}|xox[baprs]-[A-Za-z0-9-]{10,}|AKIA[0-9A-Z]{16})\b/g,
|
|
503
|
+
rewrite: () => "[REDACTED:api-key]"
|
|
504
|
+
},
|
|
505
|
+
{
|
|
506
|
+
name: "infra-host",
|
|
507
|
+
pattern: /(?<![A-Za-z0-9.-])((?:[A-Za-z0-9-]+\.)*)tangle\.(?:tools|network)(?![A-Za-z0-9-])/g,
|
|
508
|
+
rewrite: (_match, prefix) => `${prefix ?? ""}internal.example`
|
|
509
|
+
},
|
|
510
|
+
{
|
|
511
|
+
name: "machine-host",
|
|
512
|
+
pattern: /\b[A-Za-z0-9]+-GTR-Pro\b/g,
|
|
513
|
+
rewrite: () => "workstation"
|
|
514
|
+
}
|
|
515
|
+
];
|
|
516
|
+
function emptyScrubCounts() {
|
|
517
|
+
return Object.fromEntries(SCRUB_RULES.map((rule) => [rule.name, 0]));
|
|
518
|
+
}
|
|
519
|
+
function addScrubCounts(into, from) {
|
|
520
|
+
for (const [name, count] of Object.entries(from)) into[name] = (into[name] ?? 0) + count;
|
|
521
|
+
return into;
|
|
522
|
+
}
|
|
523
|
+
function scrubText(text, counts) {
|
|
524
|
+
let out = text;
|
|
525
|
+
for (const rule of SCRUB_RULES) out = out.replace(rule.pattern, (match, ...rest) => {
|
|
526
|
+
counts[rule.name] = (counts[rule.name] ?? 0) + 1;
|
|
527
|
+
return rule.rewrite(match, typeof rest[0] === "string" ? rest[0] : void 0);
|
|
528
|
+
});
|
|
529
|
+
return out;
|
|
530
|
+
}
|
|
531
|
+
function scrubValue(value, counts) {
|
|
532
|
+
if (typeof value === "string") return scrubText(value, counts);
|
|
533
|
+
if (Array.isArray(value)) return value.map((item) => scrubValue(item, counts));
|
|
534
|
+
if (value !== null && typeof value === "object") {
|
|
535
|
+
const out = {};
|
|
536
|
+
for (const [key, item] of Object.entries(value)) out[key] = scrubValue(item, counts);
|
|
537
|
+
return out;
|
|
538
|
+
}
|
|
539
|
+
return value;
|
|
540
|
+
}
|
|
541
|
+
/**
|
|
542
|
+
* Scrub every string value in a line; structure and key order are preserved.
|
|
543
|
+
*
|
|
544
|
+
* `assertMinted` on the way out rather than a cast: scrubbing rebuilds the
|
|
545
|
+
* object, so the brand has to be re-earned, and re-validating proves the rules
|
|
546
|
+
* did not rewrite a field the schema constrains (`reward` is a number, not a
|
|
547
|
+
* string, so no rule should ever touch it — this is what checks that).
|
|
548
|
+
*/
|
|
549
|
+
function scrubRolloutLine(line, counts) {
|
|
550
|
+
return assertMinted(scrubValue(line, counts), `scrubbed rollout line ${line.rollout_id}`);
|
|
551
|
+
}
|
|
552
|
+
function scrubLines(lines) {
|
|
553
|
+
const counts = emptyScrubCounts();
|
|
554
|
+
return {
|
|
555
|
+
lines: lines.map((line) => scrubRolloutLine(line, counts)),
|
|
556
|
+
counts
|
|
557
|
+
};
|
|
558
|
+
}
|
|
559
|
+
/**
|
|
560
|
+
* A `RolloutScrubber` (text → text) applying the full rule set — the
|
|
561
|
+
* default hook to pass to `mintRolloutRows({ scrub })` so lines are
|
|
562
|
+
* scrubbed at mint time, before they ever reach a ledger file. Release
|
|
563
|
+
* builds re-run `scrubLines` regardless (idempotent), so double-scrubbing
|
|
564
|
+
* is safe and counted as zero.
|
|
565
|
+
*/
|
|
566
|
+
function defaultRolloutScrubber(text) {
|
|
567
|
+
return scrubText(text, emptyScrubCounts());
|
|
568
|
+
}
|
|
569
|
+
//#endregion
|
|
570
|
+
//#region src/rollout/release/hf-dataset.ts
|
|
571
|
+
/**
|
|
572
|
+
* One-command HuggingFace dataset release from rollout ledgers:
|
|
573
|
+
*
|
|
574
|
+
* agent-eval rollout-release <ledger.jsonl...> --out <dir> \
|
|
575
|
+
* [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]
|
|
576
|
+
*
|
|
577
|
+
* Pipeline per input ledger: read + validate → fail-closed filters
|
|
578
|
+
* (trainable split only; proposer sessions dropped unless
|
|
579
|
+
* --include-proposers, they contain improvement-loop harness source) →
|
|
580
|
+
* deterministic scrub → export the requested formats + scrub-report.json +
|
|
581
|
+
* auto-generated README.md card. Deterministic: same inputs and flags →
|
|
582
|
+
* byte-identical output dir.
|
|
583
|
+
*
|
|
584
|
+
* --push uploads the built dir with `huggingface-cli upload` only when the
|
|
585
|
+
* CLI exists on PATH and HF_TOKEN is present in the env; the token is
|
|
586
|
+
* never printed. Everything else runs fully offline.
|
|
587
|
+
*/
|
|
588
|
+
async function buildHfDataset(inputs, options) {
|
|
589
|
+
if (inputs.length === 0) throw new Error("no input ledgers given");
|
|
590
|
+
if (options.formats.length === 0) throw new Error("no formats selected");
|
|
591
|
+
const report = {
|
|
592
|
+
files: {},
|
|
593
|
+
totals: emptyScrubCounts(),
|
|
594
|
+
excluded: {
|
|
595
|
+
proposers: 0,
|
|
596
|
+
nonTrain: 0
|
|
597
|
+
}
|
|
598
|
+
};
|
|
599
|
+
const kept = [];
|
|
600
|
+
let read = 0;
|
|
601
|
+
for (const input of inputs) {
|
|
602
|
+
const lines = await readRolloutLedger(input);
|
|
603
|
+
read += lines.length;
|
|
604
|
+
const scrubbed = scrubLines(lines.filter((line) => {
|
|
605
|
+
if (!isTrainableSplit(line.task.split)) {
|
|
606
|
+
report.excluded.nonTrain += 1;
|
|
607
|
+
return false;
|
|
608
|
+
}
|
|
609
|
+
if (!options.includeProposers && line.role === "proposer") {
|
|
610
|
+
report.excluded.proposers += 1;
|
|
611
|
+
return false;
|
|
612
|
+
}
|
|
613
|
+
return true;
|
|
614
|
+
}));
|
|
615
|
+
report.files[input] = scrubbed.counts;
|
|
616
|
+
addScrubCounts(report.totals, scrubbed.counts);
|
|
617
|
+
kept.push(...scrubbed.lines);
|
|
618
|
+
}
|
|
619
|
+
const formatCounts = {};
|
|
620
|
+
const files = [];
|
|
621
|
+
const gated = gatedRolloutIds(kept);
|
|
622
|
+
const gate = {
|
|
623
|
+
gatedLines: gated.size,
|
|
624
|
+
byFormat: {}
|
|
625
|
+
};
|
|
626
|
+
const pending = [];
|
|
627
|
+
for (const format of options.formats) {
|
|
628
|
+
const path = join(options.out, FORMAT_FILES[format]);
|
|
629
|
+
if (format === "raw") {
|
|
630
|
+
gate.byFormat.raw = measureFormatGate(gated, releaseRowRefs.raw(kept));
|
|
631
|
+
formatCounts.raw = kept.length;
|
|
632
|
+
pending.push({
|
|
633
|
+
path,
|
|
634
|
+
write: () => writeRolloutLedger(path, kept)
|
|
635
|
+
});
|
|
636
|
+
} else if (format === "sft") {
|
|
637
|
+
const rows = toSftRows(kept);
|
|
638
|
+
gate.byFormat.sft = measureFormatGate(gated, releaseRowRefs.sft(rows));
|
|
639
|
+
formatCounts.sft = rows.length;
|
|
640
|
+
pending.push({
|
|
641
|
+
path,
|
|
642
|
+
write: () => writeFile(path, toJsonl(rows))
|
|
643
|
+
});
|
|
644
|
+
} else if (format === "verifiers") {
|
|
645
|
+
const outputs = toVerifiersRolloutOutputs(kept, { gatedLines: FORMAT_GATE_DISPOSITION.verifiers });
|
|
646
|
+
gate.byFormat.verifiers = measureFormatGate(gated, releaseRowRefs.verifiers(outputs));
|
|
647
|
+
formatCounts.verifiers = outputs.length;
|
|
648
|
+
pending.push({
|
|
649
|
+
path,
|
|
650
|
+
write: () => writeFile(path, toJsonl(outputs))
|
|
651
|
+
});
|
|
652
|
+
} else {
|
|
653
|
+
const items = toRftItems(kept, { gatedLines: FORMAT_GATE_DISPOSITION.rft });
|
|
654
|
+
gate.byFormat.rft = measureFormatGate(gated, releaseRowRefs.rft(items));
|
|
655
|
+
formatCounts.rft = items.length;
|
|
656
|
+
pending.push({
|
|
657
|
+
path,
|
|
658
|
+
write: () => writeFile(path, toJsonl(items))
|
|
659
|
+
});
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
assertGateReport(gate);
|
|
663
|
+
for (const { path, write } of pending) {
|
|
664
|
+
await mkdir(dirname(path), { recursive: true });
|
|
665
|
+
await write();
|
|
666
|
+
files.push(path);
|
|
667
|
+
}
|
|
668
|
+
const reportPath = join(options.out, "scrub-report.json");
|
|
669
|
+
await writeFile(reportPath, `${JSON.stringify(report, null, 2)}\n`);
|
|
670
|
+
files.push(reportPath);
|
|
671
|
+
const cardPath = join(options.out, "README.md");
|
|
672
|
+
await writeFile(cardPath, buildDatasetCard({
|
|
673
|
+
lines: kept,
|
|
674
|
+
formats: options.formats,
|
|
675
|
+
includeProposers: options.includeProposers,
|
|
676
|
+
sourceFiles: inputs.map((input) => basename(input)),
|
|
677
|
+
scrubTotals: report.totals,
|
|
678
|
+
excluded: report.excluded,
|
|
679
|
+
formatCounts,
|
|
680
|
+
gate
|
|
681
|
+
}));
|
|
682
|
+
files.push(cardPath);
|
|
683
|
+
return {
|
|
684
|
+
inputs,
|
|
685
|
+
read,
|
|
686
|
+
kept: kept.length,
|
|
687
|
+
scrub: report,
|
|
688
|
+
formatCounts,
|
|
689
|
+
gate,
|
|
690
|
+
files
|
|
691
|
+
};
|
|
692
|
+
}
|
|
693
|
+
function planPushCommand(repo, outDir) {
|
|
694
|
+
return [
|
|
695
|
+
"huggingface-cli",
|
|
696
|
+
"upload",
|
|
697
|
+
repo,
|
|
698
|
+
outDir,
|
|
699
|
+
".",
|
|
700
|
+
"--repo-type",
|
|
701
|
+
"dataset"
|
|
702
|
+
];
|
|
703
|
+
}
|
|
704
|
+
function pushDataset(repo, outDir) {
|
|
705
|
+
if (spawnSync("which", ["huggingface-cli"], { stdio: "ignore" }).status !== 0) throw new Error("huggingface-cli not found on PATH — install huggingface_hub[cli] before --push");
|
|
706
|
+
if (!process.env.HF_TOKEN) throw new Error("HF_TOKEN not present in env — refusing to push");
|
|
707
|
+
const [command, ...args] = planPushCommand(repo, outDir);
|
|
708
|
+
const run = spawnSync(command, args, { stdio: "inherit" });
|
|
709
|
+
if (run.status !== 0) throw new Error(`huggingface-cli upload exited ${String(run.status)}`);
|
|
710
|
+
}
|
|
711
|
+
const ROLLOUT_RELEASE_USAGE = "usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]";
|
|
712
|
+
function parseRolloutReleaseArgs(argv) {
|
|
713
|
+
const args = {
|
|
714
|
+
inputs: [],
|
|
715
|
+
out: "",
|
|
716
|
+
formats: [...RELEASE_FORMATS],
|
|
717
|
+
includeProposers: false,
|
|
718
|
+
push: null
|
|
719
|
+
};
|
|
720
|
+
for (let i = 0; i < argv.length; i++) {
|
|
721
|
+
const arg = argv[i];
|
|
722
|
+
if (arg === "--out") args.out = argv[++i] ?? "";
|
|
723
|
+
else if (arg === "--formats") {
|
|
724
|
+
const raw = (argv[++i] ?? "").split(",").filter(Boolean);
|
|
725
|
+
for (const format of raw) if (!RELEASE_FORMATS.includes(format)) throw new Error(`unknown format "${format}" — expected one of ${RELEASE_FORMATS.join(",")}`);
|
|
726
|
+
args.formats = raw;
|
|
727
|
+
} else if (arg === "--include-proposers") args.includeProposers = true;
|
|
728
|
+
else if (arg === "--push") args.push = argv[++i] ?? null;
|
|
729
|
+
else if (arg.startsWith("--")) throw new Error(`unknown flag "${arg}"`);
|
|
730
|
+
else args.inputs.push(arg);
|
|
731
|
+
}
|
|
732
|
+
if (args.inputs.length === 0 || !args.out) throw new Error(ROLLOUT_RELEASE_USAGE);
|
|
733
|
+
if (args.push !== null && !/^[\w.-]+\/[\w.-]+$/.test(args.push)) throw new Error(`--push expects <org/name>, got "${args.push}"`);
|
|
734
|
+
return args;
|
|
735
|
+
}
|
|
736
|
+
/** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
|
|
737
|
+
async function runRolloutReleaseCli(argv) {
|
|
738
|
+
let args;
|
|
739
|
+
try {
|
|
740
|
+
args = parseRolloutReleaseArgs(argv);
|
|
741
|
+
} catch (error) {
|
|
742
|
+
process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
|
|
743
|
+
return 2;
|
|
744
|
+
}
|
|
745
|
+
const summary = await buildHfDataset(args.inputs, args);
|
|
746
|
+
process.stdout.write(`${JSON.stringify({
|
|
747
|
+
read: summary.read,
|
|
748
|
+
kept: summary.kept,
|
|
749
|
+
formatCounts: summary.formatCounts,
|
|
750
|
+
gate: summary.gate,
|
|
751
|
+
scrub: summary.scrub
|
|
752
|
+
}, null, 2)}\n`);
|
|
753
|
+
process.stdout.write(`dataset → ${args.out} (${summary.files.length} files)\n`);
|
|
754
|
+
if (args.push !== null) {
|
|
755
|
+
pushDataset(args.push, args.out);
|
|
756
|
+
process.stdout.write(`pushed → ${args.push}\n`);
|
|
757
|
+
}
|
|
758
|
+
return 0;
|
|
759
|
+
}
|
|
760
|
+
//#endregion
|
|
761
|
+
export { readRolloutJournal as C, appendRolloutLines as S, writeRolloutLedger as T, FORMAT_GATE_DISPOSITION as _, pushDataset as a, measureFormatGate as b, addScrubCounts as c, scrubLines as d, scrubRolloutLine as f, buildDatasetCard as g, RELEASE_FORMATS as h, planPushCommand as i, defaultRolloutScrubber as l, FORMAT_FILES as m, buildHfDataset as n, runRolloutReleaseCli as o, scrubText as p, parseRolloutReleaseArgs as r, SCRUB_RULES as s, ROLLOUT_RELEASE_USAGE as t, emptyScrubCounts as u, assertGateReport as v, readRolloutLedger as w, releaseRowRefs as x, gatedRolloutIds as y };
|
|
762
|
+
|
|
763
|
+
//# sourceMappingURL=hf-dataset-DBJXXoY1.js.map
|