@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
package/dist/rl.js
CHANGED
|
@@ -1,1833 +1,2444 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
} from "./
|
|
7
|
-
import {
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
} from "./
|
|
11
|
-
import {
|
|
12
|
-
|
|
13
|
-
} from "./
|
|
14
|
-
import {
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
} from "./chunk-NYLOYM6N.js";
|
|
20
|
-
import {
|
|
21
|
-
campaignCellToRunRecord
|
|
22
|
-
} from "./chunk-2MKQIFS4.js";
|
|
23
|
-
import "./chunk-PBE2LOSS.js";
|
|
24
|
-
import {
|
|
25
|
-
rubricPredictiveValidity
|
|
26
|
-
} from "./chunk-S5YLIBFX.js";
|
|
27
|
-
import {
|
|
28
|
-
evaluateInterimReleaseConfidence
|
|
29
|
-
} from "./chunk-MAZ26DC7.js";
|
|
30
|
-
import "./chunk-VLOATJQ2.js";
|
|
31
|
-
import "./chunk-DPUHNQLN.js";
|
|
32
|
-
import {
|
|
33
|
-
benjaminiHochberg,
|
|
34
|
-
wilcoxonSignedRank
|
|
35
|
-
} from "./chunk-MHELPNRP.js";
|
|
36
|
-
import {
|
|
37
|
-
observationsFromRunRecords,
|
|
38
|
-
thompsonCurriculum,
|
|
39
|
-
varianceBasedCurriculum
|
|
40
|
-
} from "./chunk-VZSRQ272.js";
|
|
41
|
-
import "./chunk-WS3NZZQQ.js";
|
|
42
|
-
import "./chunk-VI2UW6B6.js";
|
|
43
|
-
import "./chunk-TT4KNT67.js";
|
|
44
|
-
import "./chunk-PC4UYEBM.js";
|
|
45
|
-
import "./chunk-VQMK5FMP.js";
|
|
46
|
-
import {
|
|
47
|
-
runTaskScore
|
|
48
|
-
} from "./chunk-2JX3CFMB.js";
|
|
49
|
-
import "./chunk-MA6HLL3S.js";
|
|
50
|
-
import {
|
|
51
|
-
ValidationError
|
|
52
|
-
} from "./chunk-ONWEPEDO.js";
|
|
53
|
-
import "./chunk-PZ5AY32C.js";
|
|
54
|
-
|
|
55
|
-
// src/rl/adaptation-eval.ts
|
|
1
|
+
import { s as ValidationError } from "./errors-8YnH8WlF.js";
|
|
2
|
+
import { o as runTaskScore } from "./run-record-BuoE80Dq.js";
|
|
3
|
+
import { N as wilcoxonSignedRank, t as benjaminiHochberg } from "./statistics-CnnxdpOg.js";
|
|
4
|
+
import { r as observedSplitScore, s as trainingScore } from "./reward-nw2xZGZG.js";
|
|
5
|
+
import { l as assertRewardGate } from "./schema-C6DW4ZHR.js";
|
|
6
|
+
import { t as isSplitEligible } from "./exporters-q9iL-2Jf.js";
|
|
7
|
+
import { t as mintRolloutRows } from "./mint-yN2M2eh0.js";
|
|
8
|
+
import { a as InMemoryTraceStore } from "./integrity-BzRbCHzi.js";
|
|
9
|
+
import { c as campaignCellToRunRecord, i as filterDeterministicallyRewarded, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-qipEpKvY.js";
|
|
10
|
+
import { t as runEvalCampaign } from "./eval-campaign-DEm6c8ru.js";
|
|
11
|
+
import { t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
12
|
+
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-B3xmbmS1.js";
|
|
13
|
+
import { n as thompsonCurriculum, r as varianceBasedCurriculum, t as observationsFromRunRecords } from "./active-curriculum-C4mk67HP.js";
|
|
14
|
+
import { i as selfNormalizedImportanceWeighting, n as inverseProbabilityWeighting, r as offPolicyEstimateAll, t as doublyRobust } from "./off-policy-DvgzvtIx.js";
|
|
15
|
+
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-ChBKlTd_.js";
|
|
16
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
17
|
+
import { dirname } from "node:path";
|
|
18
|
+
//#region src/rl/adaptation-eval.ts
|
|
56
19
|
async function runAdaptationCurve(opts) {
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
20
|
+
const ks = opts.ks ?? [
|
|
21
|
+
0,
|
|
22
|
+
1,
|
|
23
|
+
2,
|
|
24
|
+
4,
|
|
25
|
+
8,
|
|
26
|
+
16
|
|
27
|
+
];
|
|
28
|
+
const reps = opts.reps ?? 3;
|
|
29
|
+
const passThreshold = opts.passThreshold ?? .5;
|
|
30
|
+
const sortedKs = [...ks].sort((a, b) => a - b);
|
|
31
|
+
const points = [];
|
|
32
|
+
for (const k of sortedKs) {
|
|
33
|
+
const perScenario = [];
|
|
34
|
+
const allScores = [];
|
|
35
|
+
let totalPasses = 0;
|
|
36
|
+
let totalAttempts = 0;
|
|
37
|
+
for (const scenario of opts.scenarios) {
|
|
38
|
+
const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
|
|
39
|
+
const scores = [];
|
|
40
|
+
let passes = 0;
|
|
41
|
+
for (let r = 0; r < reps; r++) {
|
|
42
|
+
const score = await opts.runner.run({
|
|
43
|
+
scenario,
|
|
44
|
+
k,
|
|
45
|
+
rep: r
|
|
46
|
+
});
|
|
47
|
+
scores.push(score);
|
|
48
|
+
if (score >= passThreshold) passes++;
|
|
49
|
+
allScores.push(score);
|
|
50
|
+
if (score >= passThreshold) totalPasses++;
|
|
51
|
+
totalAttempts++;
|
|
52
|
+
}
|
|
53
|
+
const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
|
|
54
|
+
perScenario.push({
|
|
55
|
+
scenarioId: sid,
|
|
56
|
+
meanScore: meanS,
|
|
57
|
+
passes,
|
|
58
|
+
total: scores.length
|
|
59
|
+
});
|
|
60
|
+
}
|
|
61
|
+
const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
|
|
62
|
+
const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
|
|
63
|
+
points.push({
|
|
64
|
+
k,
|
|
65
|
+
meanScore,
|
|
66
|
+
passRate: totalPasses / Math.max(1, totalAttempts),
|
|
67
|
+
std: Math.sqrt(variance),
|
|
68
|
+
n: allScores.length,
|
|
69
|
+
perScenario
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
const firstPassK = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
|
|
73
|
+
const maxK = sortedKs[sortedKs.length - 1] ?? 1;
|
|
74
|
+
let area = 0;
|
|
75
|
+
for (let i = 1; i < points.length; i++) {
|
|
76
|
+
const x1 = points[i - 1].k;
|
|
77
|
+
const x2 = points[i].k;
|
|
78
|
+
const y1 = points[i - 1].meanScore;
|
|
79
|
+
const y2 = points[i].meanScore;
|
|
80
|
+
area += (y1 + y2) / 2 * (x2 - x1);
|
|
81
|
+
}
|
|
82
|
+
return {
|
|
83
|
+
points,
|
|
84
|
+
firstPassK,
|
|
85
|
+
adaptationArea: maxK === 0 ? 0 : area / maxK
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Paired comparison of two adaptation curves. Per-k deltas with 95%
|
|
90
|
+
* bootstrap CIs (constructed from each curve's `perScenario` per-k means
|
|
91
|
+
* — the bootstrap unit is the scenario, not the rep).
|
|
92
|
+
*/
|
|
106
93
|
function compareAdaptationCurves(a, b, opts = {}) {
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
94
|
+
const conf = opts.confidence ?? .95;
|
|
95
|
+
const resamples = opts.bootstrapResamples ?? 500;
|
|
96
|
+
const rng = makeRng(opts.seed);
|
|
97
|
+
const perK = [];
|
|
98
|
+
for (const ap of a.points) {
|
|
99
|
+
const bp = b.points.find((p) => p.k === ap.k);
|
|
100
|
+
if (!bp) continue;
|
|
101
|
+
const aMeans = ap.perScenario.map((s) => s.meanScore);
|
|
102
|
+
const bMeans = bp.perScenario.map((s) => s.meanScore);
|
|
103
|
+
const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
|
|
104
|
+
const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
|
|
105
|
+
perK.push({
|
|
106
|
+
k: ap.k,
|
|
107
|
+
deltaMean: ap.meanScore - bp.meanScore,
|
|
108
|
+
aLow: aCi.low,
|
|
109
|
+
aHigh: aCi.high,
|
|
110
|
+
bLow: bCi.low,
|
|
111
|
+
bHigh: bCi.high
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
const areaDelta = a.adaptationArea - b.adaptationArea;
|
|
115
|
+
const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
|
|
116
|
+
const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
|
|
117
|
+
let verdict;
|
|
118
|
+
if (Math.abs(meanDelta) < .02 && Math.abs(areaDelta) < .02) verdict = "similar";
|
|
119
|
+
else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
|
|
120
|
+
else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
|
|
121
|
+
else verdict = "similar";
|
|
122
|
+
const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
|
|
123
|
+
return {
|
|
124
|
+
perK,
|
|
125
|
+
areaDelta,
|
|
126
|
+
firstPassKDelta,
|
|
127
|
+
verdict,
|
|
128
|
+
rationale
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
/** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
|
|
132
|
+
function firstPassK(curve, threshold = .5) {
|
|
133
|
+
return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
|
|
140
134
|
}
|
|
141
135
|
function makeRng(seed) {
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
136
|
+
if (seed === void 0) return Math.random;
|
|
137
|
+
let s = seed >>> 0;
|
|
138
|
+
return () => {
|
|
139
|
+
s = s + 1831565813 >>> 0;
|
|
140
|
+
let t = s;
|
|
141
|
+
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
142
|
+
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
143
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
144
|
+
};
|
|
151
145
|
}
|
|
152
146
|
function bootstrapMeanCi(xs, resamples, confidence, rng) {
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
147
|
+
if (xs.length < 2) return {
|
|
148
|
+
low: xs[0] ?? 0,
|
|
149
|
+
high: xs[0] ?? 0
|
|
150
|
+
};
|
|
151
|
+
const samples = new Array(resamples);
|
|
152
|
+
for (let b = 0; b < resamples; b++) {
|
|
153
|
+
let sum = 0;
|
|
154
|
+
for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
|
|
155
|
+
samples[b] = sum / xs.length;
|
|
156
|
+
}
|
|
157
|
+
samples.sort((a, b) => a - b);
|
|
158
|
+
const alpha = 1 - confidence;
|
|
159
|
+
return {
|
|
160
|
+
low: samples[Math.floor(alpha / 2 * resamples)],
|
|
161
|
+
high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
//#endregion
|
|
165
|
+
//#region src/rl/compute-curves.ts
|
|
166
|
+
/**
|
|
167
|
+
* Test-time compute scaling curves.
|
|
168
|
+
*
|
|
169
|
+
* The test-time-compute frontier paper (Snell et al. 2024) and the
|
|
170
|
+
* subsequent o1-style scaling work both show that LLM-agent capability
|
|
171
|
+
* is a function of the compute budget at inference, not just of the
|
|
172
|
+
* training run. The right way to characterize a candidate is therefore
|
|
173
|
+
* a *curve* — score at compute budgets {1×, 4×, 16×, …} — not a single
|
|
174
|
+
* point.
|
|
175
|
+
*
|
|
176
|
+
* This module ships:
|
|
177
|
+
*
|
|
178
|
+
* 1. The compute-curve harness — `runComputeCurve(runner, budgets)` —
|
|
179
|
+
* that evaluates one candidate at a sequence of compute budgets
|
|
180
|
+
* and returns the (compute, score) curve.
|
|
181
|
+
* 2. A best-of-N evaluator — `bestOfN(runner, n, scoreFn)` — the
|
|
182
|
+
* simplest test-time-compute scaling primitive: sample N
|
|
183
|
+
* independent rollouts, return the best.
|
|
184
|
+
* 3. A self-consistency evaluator — `selfConsistency(runner, n)` —
|
|
185
|
+
* the majority-vote variant of best-of-N for tasks with a small
|
|
186
|
+
* categorical answer space.
|
|
187
|
+
* 4. Pareto-frontier extraction over multiple candidates — given
|
|
188
|
+
* (candidate, compute, score) tuples, return the set of
|
|
189
|
+
* candidate-compute combinations that aren't dominated.
|
|
190
|
+
*
|
|
191
|
+
* Caveat: "compute" here is the caller's notion of a compute unit. For
|
|
192
|
+
* agent eval that's typically wall-time × parallelism, or token budget,
|
|
193
|
+
* or LLM-call count. We accept whatever the caller provides; the curve
|
|
194
|
+
* is on whatever axis they pick.
|
|
195
|
+
*/
|
|
169
196
|
async function runComputeCurve(opts) {
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
197
|
+
const points = [];
|
|
198
|
+
for (const budget of opts.budgets) {
|
|
199
|
+
const r = await opts.runAtBudget(budget);
|
|
200
|
+
points.push({
|
|
201
|
+
budgetId: budget.id,
|
|
202
|
+
cost: budget.cost,
|
|
203
|
+
score: r.score,
|
|
204
|
+
samples: r.samples,
|
|
205
|
+
std: r.std,
|
|
206
|
+
metrics: r.metrics
|
|
207
|
+
});
|
|
208
|
+
}
|
|
209
|
+
const sorted = [...points].sort((a, b) => a.cost - b.cost);
|
|
210
|
+
const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null;
|
|
211
|
+
const best = points.reduce((a, b) => b.score > a.score ? b : a);
|
|
212
|
+
return {
|
|
213
|
+
candidateId: opts.candidateId,
|
|
214
|
+
points: sorted,
|
|
215
|
+
logSlope,
|
|
216
|
+
best
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
/** The simplest test-time scaling primitive. */
|
|
187
220
|
async function bestOfN(opts) {
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
}
|
|
221
|
+
if (opts.n <= 0) throw new ValidationError("bestOfN: n must be > 0");
|
|
222
|
+
const rollouts = [];
|
|
223
|
+
const scores = [];
|
|
224
|
+
for (let i = 0; i < opts.n; i++) {
|
|
225
|
+
const r = await opts.sample(i);
|
|
226
|
+
rollouts.push(r);
|
|
227
|
+
scores.push(await opts.scoreFn(r));
|
|
228
|
+
}
|
|
229
|
+
let bestIndex = 0;
|
|
230
|
+
for (let i = 1; i < scores.length; i++) if (scores[i] > scores[bestIndex]) bestIndex = i;
|
|
231
|
+
const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length;
|
|
232
|
+
return {
|
|
233
|
+
best: rollouts[bestIndex],
|
|
234
|
+
bestScore: scores[bestIndex],
|
|
235
|
+
scores,
|
|
236
|
+
meanScore,
|
|
237
|
+
bestIndex
|
|
238
|
+
};
|
|
239
|
+
}
|
|
240
|
+
/**
|
|
241
|
+
* Self-consistency / majority-vote test-time scaling. For tasks with a
|
|
242
|
+
* small categorical answer space (math problems, multiple choice).
|
|
243
|
+
*/
|
|
207
244
|
async function selfConsistency(opts) {
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
rollouts
|
|
232
|
-
};
|
|
245
|
+
if (opts.n <= 0) throw new ValidationError("selfConsistency: n must be > 0");
|
|
246
|
+
const rollouts = [];
|
|
247
|
+
const histogram = {};
|
|
248
|
+
for (let i = 0; i < opts.n; i++) {
|
|
249
|
+
const r = await opts.sample(i);
|
|
250
|
+
rollouts.push(r);
|
|
251
|
+
const key = opts.answerKey(r);
|
|
252
|
+
histogram[key] = (histogram[key] ?? 0) + 1;
|
|
253
|
+
}
|
|
254
|
+
let answer = "";
|
|
255
|
+
let max = -1;
|
|
256
|
+
for (const [k, v] of Object.entries(histogram)) if (v > max) {
|
|
257
|
+
max = v;
|
|
258
|
+
answer = k;
|
|
259
|
+
}
|
|
260
|
+
const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0];
|
|
261
|
+
return {
|
|
262
|
+
answer,
|
|
263
|
+
agreement: max / opts.n,
|
|
264
|
+
histogram,
|
|
265
|
+
representative,
|
|
266
|
+
rollouts
|
|
267
|
+
};
|
|
233
268
|
}
|
|
234
269
|
function paretoFrontier(points) {
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
(q) => q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score)
|
|
239
|
-
);
|
|
240
|
-
if (!dominated) onFrontier.push(p);
|
|
241
|
-
}
|
|
242
|
-
return onFrontier.sort((a, b) => a.cost - b.cost);
|
|
270
|
+
const onFrontier = [];
|
|
271
|
+
for (const p of points) if (!points.some((q) => q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score))) onFrontier.push(p);
|
|
272
|
+
return onFrontier.sort((a, b) => a.cost - b.cost);
|
|
243
273
|
}
|
|
244
274
|
function fitLogSlope(points) {
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
}
|
|
258
|
-
|
|
259
|
-
|
|
275
|
+
const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)));
|
|
276
|
+
const ys = points.map((p) => p.score);
|
|
277
|
+
const n = xs.length;
|
|
278
|
+
const mx = xs.reduce((s, x) => s + x, 0) / n;
|
|
279
|
+
const my = ys.reduce((s, y) => s + y, 0) / n;
|
|
280
|
+
let num = 0;
|
|
281
|
+
let den = 0;
|
|
282
|
+
for (let i = 0; i < n; i++) {
|
|
283
|
+
num += (xs[i] - mx) * (ys[i] - my);
|
|
284
|
+
den += (xs[i] - mx) ** 2;
|
|
285
|
+
}
|
|
286
|
+
return den === 0 ? 0 : num / den;
|
|
287
|
+
}
|
|
288
|
+
//#endregion
|
|
289
|
+
//#region src/rl/contamination.ts
|
|
290
|
+
/**
|
|
291
|
+
* Contamination probe — held-out perturbation tests.
|
|
292
|
+
*
|
|
293
|
+
* The bug class: once a benchmark scenario set is published, models train
|
|
294
|
+
* on it, and your scores become invalid. SWE-Bench-Verified, GPQA, and
|
|
295
|
+
* MMLU-Pro all exist because their predecessors got contaminated within
|
|
296
|
+
* months. The right defense is to keep a held-out *perturbed* version of
|
|
297
|
+
* every scenario — same task, slightly different surface — and check
|
|
298
|
+
* whether scores diverge significantly. Genuine capability transfers; rote
|
|
299
|
+
* memorization doesn't.
|
|
300
|
+
*
|
|
301
|
+
* This module ships the probe contract:
|
|
302
|
+
*
|
|
303
|
+
* 1. A `ScenarioPerturbation` strategy type — function that produces a
|
|
304
|
+
* perturbed scenario from an original.
|
|
305
|
+
* 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs
|
|
306
|
+
* both halves and reports per-scenario score divergence + a global
|
|
307
|
+
* contamination verdict via paired Wilcoxon.
|
|
308
|
+
* 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,
|
|
309
|
+
* `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the
|
|
310
|
+
* task's structural difficulty while breaking surface memorization.
|
|
311
|
+
*
|
|
312
|
+
* The verdict is conservative: if the perturbed-vs-original score
|
|
313
|
+
* difference is statistically significant (BH-adjusted p < 0.05) AND
|
|
314
|
+
* the median drop is > 5 percentage points, we flag *contamination
|
|
315
|
+
* suspected*. False positives are possible (the perturbation might
|
|
316
|
+
* actually be harder); the default is to flag for review, not to
|
|
317
|
+
* autoreject.
|
|
318
|
+
*/
|
|
260
319
|
async function runContaminationProbe(input, opts = {}) {
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
contaminationSuspected,
|
|
318
|
-
reason,
|
|
319
|
-
n: valid.length
|
|
320
|
-
};
|
|
321
|
-
}
|
|
320
|
+
const fdr = opts.fdr ?? .05;
|
|
321
|
+
const minMedianDrop = opts.minMedianDrop ?? .05;
|
|
322
|
+
const floor = opts.scoreFloor ?? 0;
|
|
323
|
+
if (!input.perturbed && !input.perturbation) throw new ValidationError("runContaminationProbe: must supply either `perturbed` or `perturbation`.");
|
|
324
|
+
const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
|
|
325
|
+
if (perturbed.length !== input.originals.length) throw new ValidationError(`runContaminationProbe: perturbed length ${perturbed.length} ≠ originals ${input.originals.length}`);
|
|
326
|
+
const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
|
|
327
|
+
const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
|
|
328
|
+
const perScenario = input.originals.map((s, i) => ({
|
|
329
|
+
scenarioId: input.scenarioId(s),
|
|
330
|
+
originalScore: origScores[i],
|
|
331
|
+
perturbedScore: pertScores[i],
|
|
332
|
+
delta: pertScores[i] - origScores[i],
|
|
333
|
+
qValue: NaN
|
|
334
|
+
}));
|
|
335
|
+
const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
|
|
336
|
+
if (valid.length < 4) return {
|
|
337
|
+
perScenario,
|
|
338
|
+
pairedTest: {
|
|
339
|
+
w: 0,
|
|
340
|
+
p: 1
|
|
341
|
+
},
|
|
342
|
+
medianDelta: 0,
|
|
343
|
+
meanDelta: 0,
|
|
344
|
+
contaminationSuspected: false,
|
|
345
|
+
reason: `insufficient valid scenarios (n=${valid.length}, need ≥ 4)`,
|
|
346
|
+
n: valid.length
|
|
347
|
+
};
|
|
348
|
+
const pairedTest = wilcoxonSignedRank(valid.map((p) => p.originalScore), valid.map((p) => p.perturbedScore));
|
|
349
|
+
const deltas = valid.map((p) => p.delta);
|
|
350
|
+
const sortedDeltas = [...deltas].sort((a, b) => a - b);
|
|
351
|
+
const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
|
|
352
|
+
const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
|
|
353
|
+
const { qValues } = benjaminiHochberg(valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1))), fdr);
|
|
354
|
+
for (let i = 0; i < valid.length; i++) {
|
|
355
|
+
const v = valid[i];
|
|
356
|
+
const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
|
|
357
|
+
if (idx >= 0) perScenario[idx].qValue = qValues[i];
|
|
358
|
+
}
|
|
359
|
+
const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
|
|
360
|
+
return {
|
|
361
|
+
perScenario,
|
|
362
|
+
pairedTest,
|
|
363
|
+
medianDelta: median,
|
|
364
|
+
meanDelta: mean,
|
|
365
|
+
contaminationSuspected,
|
|
366
|
+
reason: contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} ≥ ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`,
|
|
367
|
+
n: valid.length
|
|
368
|
+
};
|
|
369
|
+
}
|
|
370
|
+
/**
|
|
371
|
+
* Identifier-rename perturbation for code/text scenarios. Replaces every
|
|
372
|
+
* occurrence of the listed identifiers with synthesized aliases. Use when
|
|
373
|
+
* the scenario's structural difficulty is independent of variable names
|
|
374
|
+
* (e.g. SWE-Bench-style coding tasks).
|
|
375
|
+
*/
|
|
322
376
|
function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
}
|
|
377
|
+
return {
|
|
378
|
+
kind: "rename_variables",
|
|
379
|
+
apply(scenario) {
|
|
380
|
+
let prompt = scenario.prompt;
|
|
381
|
+
identifiers.forEach((id, i) => {
|
|
382
|
+
const replacement = rename(id, i);
|
|
383
|
+
const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
|
|
384
|
+
prompt = prompt.replace(re, replacement);
|
|
385
|
+
});
|
|
386
|
+
return {
|
|
387
|
+
...scenario,
|
|
388
|
+
prompt
|
|
389
|
+
};
|
|
390
|
+
}
|
|
391
|
+
};
|
|
392
|
+
}
|
|
393
|
+
/**
|
|
394
|
+
* Order-shuffle perturbation. Reshuffles a list-shaped section of the
|
|
395
|
+
* prompt (for QA scenarios that present options A/B/C/D — answer depends
|
|
396
|
+
* on the option labels, not order). Caller provides the section extractor.
|
|
397
|
+
*/
|
|
336
398
|
function shuffleOrder(shuffleSection, seed) {
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
}
|
|
399
|
+
let s = seed >>> 0;
|
|
400
|
+
const rng = () => {
|
|
401
|
+
s = s + 1831565813 >>> 0;
|
|
402
|
+
let t = s;
|
|
403
|
+
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
404
|
+
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
405
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
406
|
+
};
|
|
407
|
+
return {
|
|
408
|
+
kind: "shuffle_order",
|
|
409
|
+
apply(scenario) {
|
|
410
|
+
const newPrompt = shuffleSection(scenario.prompt, rng);
|
|
411
|
+
return {
|
|
412
|
+
...scenario,
|
|
413
|
+
prompt: newPrompt
|
|
414
|
+
};
|
|
415
|
+
}
|
|
416
|
+
};
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* Inject-irrelevant-clause perturbation. Adds a benign sentence that
|
|
420
|
+
* shouldn't change the answer. Tests for "did the model just memorize
|
|
421
|
+
* the input string."
|
|
422
|
+
*/
|
|
353
423
|
function injectIrrelevantClause(clause, position = "prefix") {
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
424
|
+
return {
|
|
425
|
+
kind: "inject_irrelevant_clause",
|
|
426
|
+
apply(scenario) {
|
|
427
|
+
const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
|
|
428
|
+
return {
|
|
429
|
+
...scenario,
|
|
430
|
+
prompt
|
|
431
|
+
};
|
|
432
|
+
}
|
|
433
|
+
};
|
|
361
434
|
}
|
|
362
435
|
function escapeRegex(s) {
|
|
363
|
-
|
|
364
|
-
}
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
436
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
437
|
+
}
|
|
438
|
+
//#endregion
|
|
439
|
+
//#region src/rl/rollout-input.ts
|
|
440
|
+
/**
|
|
441
|
+
* Shared checks for trainer exports over canonical minted rollout lines.
|
|
442
|
+
*
|
|
443
|
+
* Exporters accept only `MintedRolloutLine[]`. Callers convert run records with
|
|
444
|
+
* `mintRolloutRows` before deriving preferences or trainer files.
|
|
445
|
+
*/
|
|
446
|
+
/**
|
|
447
|
+
* The line's reward, with `null` meaning "no verdict exists" and never "scored
|
|
448
|
+
* zero".
|
|
449
|
+
*
|
|
450
|
+
* The wire contract represents an absent verdict directly as `reward: null`.
|
|
451
|
+
*/
|
|
452
|
+
function trainableLineReward(line) {
|
|
453
|
+
assertRewardGate(line, "trainable reward");
|
|
454
|
+
const { reward } = line.outcome;
|
|
455
|
+
if (reward === null || !Number.isFinite(reward)) return null;
|
|
456
|
+
return reward;
|
|
457
|
+
}
|
|
458
|
+
/** The anti-Goodhart flag as it travels on the line. */
|
|
459
|
+
function isLineRealnessGated(line) {
|
|
460
|
+
return line.outcome.realness_gated === true;
|
|
461
|
+
}
|
|
462
|
+
function push(index, key, line) {
|
|
463
|
+
const existing = index.get(key);
|
|
464
|
+
if (existing === void 0) index.set(key, [line]);
|
|
465
|
+
else existing.push(line);
|
|
466
|
+
}
|
|
467
|
+
function invocationIndex(lines) {
|
|
468
|
+
const byRollout = /* @__PURE__ */ new Map();
|
|
469
|
+
const byRun = /* @__PURE__ */ new Map();
|
|
470
|
+
for (const line of lines) {
|
|
471
|
+
push(byRollout, line.rollout_id, line);
|
|
472
|
+
push(byRun, line.run_id, line);
|
|
473
|
+
}
|
|
474
|
+
return {
|
|
475
|
+
byRollout,
|
|
476
|
+
byRun
|
|
477
|
+
};
|
|
478
|
+
}
|
|
479
|
+
/**
|
|
480
|
+
* Resolve one referenced id to exactly ONE invocation, or refuse to guess.
|
|
481
|
+
*
|
|
482
|
+
* The previous implementation was `new Map(lines.map((l) => [l.run_id, l]))`,
|
|
483
|
+
* which is LAST-WINS: with a gated supervisor node and an ungated worker sharing
|
|
484
|
+
* a `run_id`, the answer depended on which one appeared later in the array, so
|
|
485
|
+
* `[gatedRoot, worker, rival]` emitted the gamed trajectory as `chosen` and
|
|
486
|
+
* simply reordering the same input suppressed it. An order-dependent security
|
|
487
|
+
* property passes every test whose fixture happens to be ordered favourably,
|
|
488
|
+
* which is the worst possible failure mode for a gate.
|
|
489
|
+
*
|
|
490
|
+
* The rule that removes order from the answer: an id resolves only when it names
|
|
491
|
+
* one invocation. `rollout_id` is tried first because it IS the invocation id;
|
|
492
|
+
* `run_id` is accepted only when the run holds a single invocation, and a
|
|
493
|
+
* cross-index disagreement (an id that is one line's `rollout_id` and a
|
|
494
|
+
* different line's `run_id`) is ambiguous rather than silently preferring
|
|
495
|
+
* either.
|
|
496
|
+
*/
|
|
497
|
+
function resolveInvocation(index, id) {
|
|
498
|
+
const rollouts = index.byRollout.get(id) ?? [];
|
|
499
|
+
const runs = index.byRun.get(id) ?? [];
|
|
500
|
+
if (rollouts.length > 1) return {
|
|
501
|
+
kind: "ambiguous",
|
|
502
|
+
count: rollouts.length
|
|
503
|
+
};
|
|
504
|
+
const exact = rollouts[0];
|
|
505
|
+
if (exact !== void 0) {
|
|
506
|
+
if (runs.some((line) => line !== exact)) return {
|
|
507
|
+
kind: "ambiguous",
|
|
508
|
+
count: 1 + runs.filter((line) => line !== exact).length
|
|
509
|
+
};
|
|
510
|
+
return {
|
|
511
|
+
kind: "resolved",
|
|
512
|
+
line: exact
|
|
513
|
+
};
|
|
514
|
+
}
|
|
515
|
+
if (runs.length > 1) return {
|
|
516
|
+
kind: "ambiguous",
|
|
517
|
+
count: runs.length
|
|
518
|
+
};
|
|
519
|
+
const only = runs[0];
|
|
520
|
+
return only === void 0 ? { kind: "missing" } : {
|
|
521
|
+
kind: "resolved",
|
|
522
|
+
line: only
|
|
523
|
+
};
|
|
524
|
+
}
|
|
525
|
+
/**
|
|
526
|
+
* THE admission rule for every exporter whose input is line-less — one
|
|
527
|
+
* implementation, because two siblings over the same input class with different
|
|
528
|
+
* gating is the defect being eliminated, and it has now happened twice
|
|
529
|
+
* (`toPrmRows` hardened while `toDpoRows` was left open; `toGrpoRows`'
|
|
530
|
+
* `rewardOf` gated while `extractPreferences`' identically-named hook was not).
|
|
531
|
+
*
|
|
532
|
+
* Fail-closed in five steps:
|
|
533
|
+
* 1. No context at all → throw. A two-argument call used to be accepted and
|
|
534
|
+
* produced rows with no gate applied whatsoever.
|
|
535
|
+
* 2. A referenced id with NO line → throw. Its gate status is unknown, and
|
|
536
|
+
* unknown is not clean. Thrown rather than dropped because it means the
|
|
537
|
+
* caller did not supply the context it was asked for, which is a defect in
|
|
538
|
+
* the call, not in the data.
|
|
539
|
+
* 3. A referenced id naming MORE THAN ONE invocation → DROP the item and count
|
|
540
|
+
* it. Dropped rather than thrown because, unlike (2), this is ordinary data
|
|
541
|
+
* — a supervision episode legitimately holds a supervisor invocation and
|
|
542
|
+
* several workers under one `run_id` — and throwing would make these
|
|
543
|
+
* exporters unusable on any supervisor corpus, whose only workaround is for
|
|
544
|
+
* the caller to hand-filter `context.lines` down to one line per run. That
|
|
545
|
+
* workaround IS the leak, performed by hand. The count is surfaced by
|
|
546
|
+
* `admitUngatedByInvocation` so the drop is never silent.
|
|
547
|
+
* 4. Every resolved line goes through `assertRewardGate`, so the line-less
|
|
548
|
+
* exporters compose the same check list as the waist exporters instead of
|
|
549
|
+
* relying on `realness_gated` alone (which is one of three checks).
|
|
550
|
+
* 5. Either side realness-gated → DROP the item. Dropped rather than zeroed
|
|
551
|
+
* because these shapes have no honest zero: a preference pair is a
|
|
552
|
+
* statement that one trajectory is better than another, and a gamed
|
|
553
|
+
* trajectory belongs on neither side of it — as the chosen one it teaches
|
|
554
|
+
* the gaming move outright, and as the rejected one it still ships the
|
|
555
|
+
* gaming trajectory's text into the training file as a contrast example
|
|
556
|
+
* nobody asked for.
|
|
557
|
+
*
|
|
558
|
+
* `inspect` is the per-exporter extra check (PRM's trajectory-completeness
|
|
559
|
+
* rules). It runs on every resolved line before any item is admitted, so the
|
|
560
|
+
* whole batch fails before a single row is built.
|
|
561
|
+
*
|
|
562
|
+
* Pure: it reports what it dropped and prints nothing.
|
|
563
|
+
*/
|
|
564
|
+
function auditInvocationAdmission(items, idsOf, context, requirement, inspect) {
|
|
565
|
+
if (context === void 0 || context === null) throw new Error(`${requirement.exporter}: a ${requirement.contextType} is required — ${requirement.because} Pass \`{ lines: (await mintRolloutRows(...)).rows }\`.`);
|
|
566
|
+
const index = invocationIndex(context.lines);
|
|
567
|
+
const audit = {
|
|
568
|
+
admitted: [],
|
|
569
|
+
gatedDrops: 0,
|
|
570
|
+
ambiguousDrops: 0,
|
|
571
|
+
ambiguous: []
|
|
572
|
+
};
|
|
573
|
+
for (const item of items) {
|
|
574
|
+
const lines = [];
|
|
575
|
+
let ambiguous = false;
|
|
576
|
+
for (const id of idsOf(item)) {
|
|
577
|
+
const resolution = resolveInvocation(index, id);
|
|
578
|
+
if (resolution.kind === "missing") throw new Error(`${requirement.exporter}: no rollout line supplied for run ${id} — its realness gate and capture quality are unknown`);
|
|
579
|
+
if (resolution.kind === "ambiguous") {
|
|
580
|
+
ambiguous = true;
|
|
581
|
+
if (!audit.ambiguous.some((entry) => entry.id === id)) audit.ambiguous.push({
|
|
582
|
+
id,
|
|
583
|
+
invocations: resolution.count
|
|
584
|
+
});
|
|
585
|
+
continue;
|
|
586
|
+
}
|
|
587
|
+
lines.push(resolution.line);
|
|
588
|
+
}
|
|
589
|
+
for (const line of lines) {
|
|
590
|
+
assertRewardGate(line, requirement.exporter);
|
|
591
|
+
inspect?.(line);
|
|
592
|
+
}
|
|
593
|
+
if (ambiguous) {
|
|
594
|
+
audit.ambiguousDrops++;
|
|
595
|
+
continue;
|
|
596
|
+
}
|
|
597
|
+
if (lines.some(isLineRealnessGated)) {
|
|
598
|
+
audit.gatedDrops++;
|
|
599
|
+
continue;
|
|
600
|
+
}
|
|
601
|
+
audit.admitted.push(item);
|
|
602
|
+
}
|
|
603
|
+
return audit;
|
|
604
|
+
}
|
|
605
|
+
/**
|
|
606
|
+
* `auditInvocationAdmission` for the exporters, which return rows and have
|
|
607
|
+
* nowhere to put a count.
|
|
608
|
+
*
|
|
609
|
+
* The ambiguous drops are announced rather than swallowed: a caller who asked
|
|
610
|
+
* for N pairs and silently received N-k has no way to notice that a chunk of
|
|
611
|
+
* their preference data quietly evaporated, and "the training set got smaller
|
|
612
|
+
* for a reason nobody printed" is the same class of invisible failure as the
|
|
613
|
+
* gate that never ran. A gated drop is NOT announced — that one is the gate
|
|
614
|
+
* doing exactly its job, on the population the caller already knows is flagged.
|
|
615
|
+
*/
|
|
616
|
+
function admitUngatedByInvocation(items, idsOf, context, requirement, inspect) {
|
|
617
|
+
const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect);
|
|
618
|
+
if (audit.ambiguousDrops > 0) {
|
|
619
|
+
const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(", ");
|
|
620
|
+
console.warn(`[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more than one invocation in the supplied lines, so the realness gate cannot be read for the invocation the artifact meant. Reference the \`rollout_id\` instead of the \`run_id\`, or supply a context holding one invocation per run.`);
|
|
621
|
+
}
|
|
622
|
+
return audit.admitted;
|
|
623
|
+
}
|
|
624
|
+
//#endregion
|
|
625
|
+
//#region src/rl/exporters.ts
|
|
626
|
+
/**
|
|
627
|
+
* Trainer-format exporters.
|
|
628
|
+
*
|
|
629
|
+
* agent-eval produces canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`,
|
|
630
|
+
* `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
|
|
631
|
+
* different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
|
|
632
|
+
* fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
|
|
633
|
+
* JSONL conventions. Rather than ship N adapters, this module ships the
|
|
634
|
+
* canonical formats most production pipelines accept and ergonomic helpers
|
|
635
|
+
* for the rest.
|
|
636
|
+
*
|
|
637
|
+
* Shapes:
|
|
638
|
+
* - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed
|
|
639
|
+
* by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.
|
|
640
|
+
* - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.
|
|
641
|
+
* Consumed by prime-rl GRPO, verl, OpenRLHF.
|
|
642
|
+
* - **SFT** — `{messages[]}` JSONL with chosen completion as the final
|
|
643
|
+
* assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,
|
|
644
|
+
* Anthropic finetuning.
|
|
645
|
+
* - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.
|
|
646
|
+
* Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
|
|
647
|
+
*
|
|
648
|
+
* Why ship this in agent-eval rather than a separate adapter package: the
|
|
649
|
+
* canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`, etc.) are
|
|
650
|
+
* agent-eval's contract; without first-party exporters consumers reverse-
|
|
651
|
+
* engineer the mapping every release. The exporters codify it.
|
|
652
|
+
*
|
|
653
|
+
* The exporters take callbacks for any field that isn't on the canonical
|
|
654
|
+
* artifact (specifically: prompt + completion text, since the package
|
|
655
|
+
* stores only their hashes by design — full text is the consumer's
|
|
656
|
+
* trace store / raw event log).
|
|
657
|
+
*
|
|
658
|
+
* Every exporter that produces a training row accepts canonical minted rollout
|
|
659
|
+
* lines. Convert run records once with `mintRolloutRows`; downstream transforms
|
|
660
|
+
* then share one reward, split, and authenticity contract.
|
|
661
|
+
*/
|
|
662
|
+
const DPO_CONTEXT_REQUIREMENT = {
|
|
663
|
+
exporter: "DPO export",
|
|
664
|
+
contextType: "DpoLineContext",
|
|
665
|
+
because: "a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair — which is DPO trained to PREFER the gaming trajectory."
|
|
666
|
+
};
|
|
667
|
+
/**
|
|
668
|
+
* Convert preference triples to TRL-compatible DPO rows. The shape
|
|
669
|
+
* `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
|
|
670
|
+
* entry; every major DPO trainer accepts it.
|
|
671
|
+
*
|
|
672
|
+
* `context` is REQUIRED, and for the same reason it is required on the sibling
|
|
673
|
+
* `toPrmRows`: a triple is a line-less artifact. It names two run ids and a
|
|
674
|
+
* margin, and nothing on it says whether either run was flagged as gamed —
|
|
675
|
+
* so a two-argument call applied NO gate at all and emitted the row verbatim,
|
|
676
|
+
* reachable straight through the published bundle builder
|
|
677
|
+
* (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).
|
|
678
|
+
* Triples whose chosen or rejected side is realness-gated are dropped; a triple
|
|
679
|
+
* naming a run with no supplied line is refused. See `admitUngatedByInvocation` for
|
|
680
|
+
* why dropping, not zeroing, is the right disposition for a preference pair.
|
|
681
|
+
*/
|
|
682
|
+
async function toDpoRows(triples, lookups, context) {
|
|
683
|
+
const admitted = admitUngatedByInvocation(triples, (t) => [t.chosenRunId, t.rejectedRunId], context, DPO_CONTEXT_REQUIREMENT);
|
|
684
|
+
const out = [];
|
|
685
|
+
for (const t of admitted) {
|
|
686
|
+
const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
|
|
687
|
+
Promise.resolve(lookups.promptOf(t.chosenRunId)),
|
|
688
|
+
Promise.resolve(lookups.promptOf(t.rejectedRunId)),
|
|
689
|
+
Promise.resolve(lookups.completionOf(t.chosenRunId)),
|
|
690
|
+
Promise.resolve(lookups.completionOf(t.rejectedRunId))
|
|
691
|
+
]);
|
|
692
|
+
if (chosenPrompt !== rejectedPrompt) throw new Error(`toDpoRows: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`);
|
|
693
|
+
out.push({
|
|
694
|
+
prompt: chosenPrompt,
|
|
695
|
+
chosen,
|
|
696
|
+
rejected,
|
|
697
|
+
margin: t.marginScore,
|
|
698
|
+
meta: {
|
|
699
|
+
scenarioId: t.scenarioId,
|
|
700
|
+
chosenVariantId: t.chosenVariantId,
|
|
701
|
+
rejectedVariantId: t.rejectedVariantId,
|
|
702
|
+
chosenRunId: t.chosenRunId,
|
|
703
|
+
rejectedRunId: t.rejectedRunId,
|
|
704
|
+
chosenModel: t.meta.chosenModel,
|
|
705
|
+
rejectedModel: t.meta.rejectedModel
|
|
706
|
+
}
|
|
707
|
+
});
|
|
708
|
+
}
|
|
709
|
+
return out;
|
|
710
|
+
}
|
|
711
|
+
/** Serialize DPO rows as JSONL. One line per row. */
|
|
403
712
|
function toDpoJsonl(rows) {
|
|
404
|
-
|
|
405
|
-
}
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
713
|
+
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
714
|
+
}
|
|
715
|
+
/**
|
|
716
|
+
* Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —
|
|
717
|
+
* one row per scenario, with one completion per rollout on that scenario.
|
|
718
|
+
* A scenario with fewer than two rewarded completions emits no row because a
|
|
719
|
+
* group of one has no relative baseline.
|
|
720
|
+
*
|
|
721
|
+
* GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
|
|
722
|
+
* within a group of completions for the same prompt; this is the
|
|
723
|
+
* canonical input format. That relative baseline is exactly why the gate has
|
|
724
|
+
* to hold here: one gamed sibling exporting at full reward shifts the advantage
|
|
725
|
+
* of every honest run beside it.
|
|
726
|
+
*
|
|
727
|
+
* On the line path a realness-gated line stays in its group at reward 0 rather
|
|
728
|
+
* than being dropped. 0 is the honest label for a faked success and is usable
|
|
729
|
+
* signal; removing the line would also move the group's baseline, just in the
|
|
730
|
+
* other direction. (SFT differs — see `toSftRows`.)
|
|
731
|
+
*/
|
|
732
|
+
async function toGrpoRows(lines, lookups) {
|
|
733
|
+
return grpoRowsFromLines(lines, lookups);
|
|
734
|
+
}
|
|
735
|
+
async function grpoRowsFromLines(lines, lookups) {
|
|
736
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
737
|
+
for (const line of lines) {
|
|
738
|
+
if (!isSelectedSplit(line, lookups)) continue;
|
|
739
|
+
const arr = grouped.get(line.task.instance_id) ?? [];
|
|
740
|
+
arr.push(line);
|
|
741
|
+
grouped.set(line.task.instance_id, arr);
|
|
742
|
+
}
|
|
743
|
+
const rows = [];
|
|
744
|
+
for (const [scenarioId, group] of grouped.entries()) {
|
|
745
|
+
if (group.length === 0) continue;
|
|
746
|
+
const scored = [];
|
|
747
|
+
for (const line of group) {
|
|
748
|
+
const reward = trainableLineReward(line);
|
|
749
|
+
if (reward === null) continue;
|
|
750
|
+
scored.push({
|
|
751
|
+
line,
|
|
752
|
+
reward
|
|
753
|
+
});
|
|
754
|
+
}
|
|
755
|
+
if (scored.length < 2) continue;
|
|
756
|
+
const prompts = await Promise.all(scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id))));
|
|
757
|
+
const prompt = prompts[0];
|
|
758
|
+
if (prompts.some((value) => value !== prompt)) throw new Error(`toGrpoRows: scenario "${scenarioId}" resolves to different prompt text within one group`);
|
|
759
|
+
const completions = await Promise.all(scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id))));
|
|
760
|
+
const rewards = scored.map(({ reward }) => reward);
|
|
761
|
+
const runIds = scored.map(({ line }) => line.run_id);
|
|
762
|
+
rows.push({
|
|
763
|
+
prompt,
|
|
764
|
+
completions,
|
|
765
|
+
rewards,
|
|
766
|
+
runIds,
|
|
767
|
+
meta: {
|
|
768
|
+
scenarioId,
|
|
769
|
+
n: completions.length,
|
|
770
|
+
meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
|
|
771
|
+
}
|
|
772
|
+
});
|
|
773
|
+
}
|
|
774
|
+
return rows;
|
|
460
775
|
}
|
|
461
776
|
function toGrpoJsonl(rows) {
|
|
462
|
-
|
|
463
|
-
}
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
777
|
+
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
778
|
+
}
|
|
779
|
+
/**
|
|
780
|
+
* Convert rollout lines into Hugging Face / OpenAI / Anthropic-style
|
|
781
|
+
* conversational SFT rows. By default every qualifying line becomes one row;
|
|
782
|
+
* pass `include` to filter further (e.g., keep only `reward >= 0.8` for
|
|
783
|
+
* rejection-sampling SFT).
|
|
784
|
+
*
|
|
785
|
+
* Realness-gated lines are dropped outright, not zeroed. SFT is imitation
|
|
786
|
+
* learning: unlike GRPO, where a 0 reward teaches "this trajectory was bad",
|
|
787
|
+
* every row here is a target to copy, so a gamed trajectory must not be in the
|
|
788
|
+
* file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.
|
|
789
|
+
*
|
|
790
|
+
* The exporter is fail-closed on the split, same rule as
|
|
791
|
+
* `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by
|
|
792
|
+
* default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and
|
|
793
|
+
* `canary` never pass the default rule. A non-training bundle that wants an
|
|
794
|
+
* explicit slice (e.g. a holdout-only eval bundle) names it with
|
|
795
|
+
* `splitFilter: ['holdout']` — explicit selection replaces the default rule.
|
|
796
|
+
*/
|
|
797
|
+
async function toSftRows(lines, lookups) {
|
|
798
|
+
return sftRowsFromLines(lines, lookups);
|
|
799
|
+
}
|
|
800
|
+
async function sftRowsFromLines(lines, lookups) {
|
|
801
|
+
const include = lookups.include ?? (() => true);
|
|
802
|
+
const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0;
|
|
803
|
+
if (!Number.isFinite(minimumQualityExclusive)) throw new Error("minimumQualityExclusive must be finite");
|
|
804
|
+
const rows = [];
|
|
805
|
+
for (const line of lines) {
|
|
806
|
+
assertRewardGate(line, "SFT export");
|
|
807
|
+
if (isLineRealnessGated(line)) continue;
|
|
808
|
+
if (!isSelectedSplit(line, lookups)) continue;
|
|
809
|
+
const score = trainableLineReward(line);
|
|
810
|
+
if (score === null || score <= minimumQualityExclusive) continue;
|
|
811
|
+
if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) continue;
|
|
812
|
+
if (!include(line)) continue;
|
|
813
|
+
const system = lookups.systemOf?.(line);
|
|
814
|
+
const [prompt, completion] = await Promise.all([Promise.resolve(lookups.promptOf(line.run_id)), Promise.resolve(lookups.completionOf(line.run_id))]);
|
|
815
|
+
const messages = [];
|
|
816
|
+
if (system) messages.push({
|
|
817
|
+
role: "system",
|
|
818
|
+
content: system
|
|
819
|
+
});
|
|
820
|
+
messages.push({
|
|
821
|
+
role: "user",
|
|
822
|
+
content: prompt
|
|
823
|
+
});
|
|
824
|
+
messages.push({
|
|
825
|
+
role: "assistant",
|
|
826
|
+
content: completion
|
|
827
|
+
});
|
|
828
|
+
rows.push({
|
|
829
|
+
messages,
|
|
830
|
+
meta: {
|
|
831
|
+
runId: line.run_id,
|
|
832
|
+
candidateId: line.candidate_id ?? null,
|
|
833
|
+
scenarioId: line.task.instance_id,
|
|
834
|
+
score,
|
|
835
|
+
model: line.policy.model
|
|
836
|
+
}
|
|
837
|
+
});
|
|
838
|
+
}
|
|
839
|
+
return rows;
|
|
492
840
|
}
|
|
493
841
|
function toSftJsonl(rows) {
|
|
494
|
-
|
|
495
|
-
}
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
842
|
+
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
843
|
+
}
|
|
844
|
+
/**
|
|
845
|
+
* Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
|
|
846
|
+
* callback resolves span text from the consumer's trace store.
|
|
847
|
+
*
|
|
848
|
+
* Every referenced run is checked against its minted line before any row is
|
|
849
|
+
* emitted, and the export FAILS LOUD on a trajectory that was never fully
|
|
850
|
+
* captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected
|
|
851
|
+
* side is realness-gated are dropped instead: a capture defect is the caller's
|
|
852
|
+
* mint configuration and must be fixed, whereas a gamed run is exactly the
|
|
853
|
+
* condition the gate exists to filter.
|
|
854
|
+
*
|
|
855
|
+
* `context` is REQUIRED. A two-argument call used to be accepted and produced
|
|
856
|
+
* rows with no gate applied at all — a `PrmTrainingTriple` carries a bare
|
|
857
|
+
* `chosenReward` number and nothing that says which run it came from is honest,
|
|
858
|
+
* so with no lines this exporter has no way to learn that its chosen step is a
|
|
859
|
+
* step from a run that faked its success. It now throws: fail closed, because
|
|
860
|
+
* the alternative is a process-reward model taught to prefer the gaming move at
|
|
861
|
+
* the exact step the gaming happened.
|
|
862
|
+
*/
|
|
863
|
+
async function toPrmRows(triples, lookups, context) {
|
|
864
|
+
const admitted = admitPrmTriples(triples, context);
|
|
865
|
+
const rows = [];
|
|
866
|
+
for (const t of admitted) {
|
|
867
|
+
const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
|
|
868
|
+
const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
|
|
869
|
+
const prefixStepText = [];
|
|
870
|
+
for (const spanId of prefixSpanIds) prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)));
|
|
871
|
+
const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId));
|
|
872
|
+
const rejectedStep = await Promise.resolve(lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId));
|
|
873
|
+
rows.push({
|
|
874
|
+
prompt,
|
|
875
|
+
prefixSpanIds,
|
|
876
|
+
prefixStepText,
|
|
877
|
+
chosenStep,
|
|
878
|
+
rejectedStep,
|
|
879
|
+
chosenReward: t.chosenReward,
|
|
880
|
+
rejectedReward: t.rejectedReward,
|
|
881
|
+
marginScore: t.marginScore,
|
|
882
|
+
meta: {
|
|
883
|
+
prefixRunId: t.prefixRunId,
|
|
884
|
+
rejectedRunId: t.rejectedRunId,
|
|
885
|
+
prefixStepIndex: t.prefixStepIndex
|
|
886
|
+
}
|
|
887
|
+
});
|
|
888
|
+
}
|
|
889
|
+
return rows;
|
|
890
|
+
}
|
|
891
|
+
/**
|
|
892
|
+
* Refuse to build a process-reward row from a trajectory we do not fully have.
|
|
893
|
+
*
|
|
894
|
+
* PRM training assigns credit step by step, so a missing or silently shortened
|
|
895
|
+
* step list is not degraded data — it is data about a trajectory that never
|
|
896
|
+
* existed. Every condition below throws rather than filters, because each one
|
|
897
|
+
* means the CALLER's capture or mint configuration is wrong.
|
|
898
|
+
*/
|
|
899
|
+
function assertPrmTrainableLine(line, mintedWithMaxSteps) {
|
|
900
|
+
const id = line.rollout_id;
|
|
901
|
+
if (line.provenance.gap !== void 0) throw new Error(`PRM export: rollout ${id} is a gap line (${line.provenance.gap}) — refusing to build a process-reward row from a trajectory that was never captured`);
|
|
902
|
+
if (line.steps === void 0 || line.steps.length === 0) throw new Error(`PRM export: rollout ${id} carries no steps — refusing to build a process-reward row with no trajectory`);
|
|
903
|
+
if (line.outcome.is_truncated) throw new Error(`PRM export: rollout ${id} is marked truncated — refusing to assign step-level credit over a partial trajectory`);
|
|
904
|
+
if (mintedWithMaxSteps !== void 0 && line.steps.length >= mintedWithMaxSteps) throw new Error(`PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} — its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`);
|
|
905
|
+
}
|
|
906
|
+
const PRM_CONTEXT_REQUIREMENT = {
|
|
907
|
+
exporter: "PRM export",
|
|
908
|
+
contextType: "PrmLineContext",
|
|
909
|
+
because: "without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one."
|
|
910
|
+
};
|
|
911
|
+
/**
|
|
912
|
+
* Validate every referenced line up front (fail loud, before a single row is
|
|
913
|
+
* written) and then drop the triples whose evidence is realness-gated.
|
|
914
|
+
*
|
|
915
|
+
* The gate half is `admitUngatedByInvocation`, shared with `toDpoRows` and
|
|
916
|
+
* `stepRewardsToJsonl`; only the trajectory-completeness rules are PRM's own.
|
|
917
|
+
*/
|
|
918
|
+
function admitPrmTriples(triples, context) {
|
|
919
|
+
return admitUngatedByInvocation(triples, (t) => [t.prefixRunId, t.rejectedRunId], context, PRM_CONTEXT_REQUIREMENT, (line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps));
|
|
526
920
|
}
|
|
527
921
|
function toPrmJsonl(rows) {
|
|
528
|
-
|
|
529
|
-
}
|
|
530
|
-
function stepRewardsToJsonl(stepRewards) {
|
|
531
|
-
const rows = stepRewards.map((s) => ({
|
|
532
|
-
runId: s.runId,
|
|
533
|
-
spanId: s.spanId,
|
|
534
|
-
stepIndex: s.stepIndex,
|
|
535
|
-
reward: s.reward,
|
|
536
|
-
determinism: s.determinism,
|
|
537
|
-
weight: s.weight ?? 1
|
|
538
|
-
}));
|
|
539
|
-
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
540
|
-
}
|
|
541
|
-
function defaultReward(run) {
|
|
542
|
-
return runTaskScore(run) ?? null;
|
|
543
|
-
}
|
|
544
|
-
function isTrainingRunEligible(run, quality, options = {}) {
|
|
545
|
-
const minimumQualityExclusive = options.minimumQualityExclusive ?? 0;
|
|
546
|
-
if (!Number.isFinite(minimumQualityExclusive)) {
|
|
547
|
-
throw new Error("minimumQualityExclusive must be finite");
|
|
548
|
-
}
|
|
549
|
-
if (quality === null || quality === void 0) return false;
|
|
550
|
-
if (!Number.isFinite(quality)) {
|
|
551
|
-
throw new Error(`training quality for run "${run.runId}" must be finite`);
|
|
552
|
-
}
|
|
553
|
-
if (quality <= minimumQualityExclusive) return false;
|
|
554
|
-
if (run.terminalOutcome !== "succeeded") return false;
|
|
555
|
-
if (run.failureClass !== void 0 || run.terminalFailureReason !== void 0) return false;
|
|
556
|
-
if (run.outcome.realness?.gated === true) return false;
|
|
557
|
-
if (run.splitTag === "search") return true;
|
|
558
|
-
return run.splitTag === "holdout" && options.allowHeldOutTrainingData === true;
|
|
922
|
+
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
559
923
|
}
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
924
|
+
const STEP_REWARD_CONTEXT_REQUIREMENT = {
|
|
925
|
+
exporter: "step-reward export",
|
|
926
|
+
contextType: "RolloutLineContext",
|
|
927
|
+
because: "a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success — so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere."
|
|
928
|
+
};
|
|
929
|
+
/**
|
|
930
|
+
* Step-level reward rows as JSONL.
|
|
931
|
+
*
|
|
932
|
+
* `context` is REQUIRED for the same reason it is on `toDpoRows` and
|
|
933
|
+
* `toPrmRows`: this is a line-less input carrying a reward number. Steps
|
|
934
|
+
* belonging to a realness-gated run are dropped rather than zeroed — a
|
|
935
|
+
* per-step reward of 0 across a whole trajectory is a claim that every step was
|
|
936
|
+
* bad, which is a different (and false) statement from "this run's success was
|
|
937
|
+
* fabricated, so its step-level credit assignment is meaningless".
|
|
938
|
+
*/
|
|
939
|
+
function stepRewardsToJsonl(stepRewards, context) {
|
|
940
|
+
const rows = admitUngatedByInvocation(stepRewards, (s) => [s.runId], context, STEP_REWARD_CONTEXT_REQUIREMENT).map((s) => ({
|
|
941
|
+
runId: s.runId,
|
|
942
|
+
spanId: s.spanId,
|
|
943
|
+
stepIndex: s.stepIndex,
|
|
944
|
+
reward: s.reward,
|
|
945
|
+
determinism: s.determinism,
|
|
946
|
+
weight: s.weight ?? 1
|
|
947
|
+
}));
|
|
948
|
+
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
949
|
+
}
|
|
950
|
+
function isSelectedSplit(line, options) {
|
|
951
|
+
if (options.splitFilter !== void 0) return options.splitFilter.includes(line.task.split);
|
|
952
|
+
return isSplitEligible(line, options);
|
|
953
|
+
}
|
|
954
|
+
//#endregion
|
|
955
|
+
//#region src/rl/dataset.ts
|
|
956
|
+
const DATASET_FORMAT_SET = /* @__PURE__ */ new Set([
|
|
957
|
+
"grpo",
|
|
958
|
+
"sft",
|
|
959
|
+
"dpo"
|
|
960
|
+
]);
|
|
564
961
|
function validateDatasetFormats(value) {
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
const datasetFormat = format;
|
|
577
|
-
if (seen.has(datasetFormat)) {
|
|
578
|
-
throw new Error(
|
|
579
|
-
`buildRlDataset: duplicate format ${JSON.stringify(datasetFormat)}; each format may be requested once`
|
|
580
|
-
);
|
|
581
|
-
}
|
|
582
|
-
seen.add(datasetFormat);
|
|
583
|
-
formats.push(datasetFormat);
|
|
584
|
-
}
|
|
585
|
-
return formats;
|
|
586
|
-
}
|
|
587
|
-
function reward(r, rewardOf2) {
|
|
588
|
-
const value = rewardOf2 ? rewardOf2(r) : runTaskScore(r) ?? null;
|
|
589
|
-
if (value !== null && !Number.isFinite(value)) {
|
|
590
|
-
throw new Error(`buildRlDataset: reward for run "${r.runId}" must be finite`);
|
|
591
|
-
}
|
|
592
|
-
return value;
|
|
962
|
+
if (!Array.isArray(value) || value.length === 0) throw new Error("buildRlDataset: formats must contain at least one of: grpo, sft, dpo");
|
|
963
|
+
const formats = [];
|
|
964
|
+
const seen = /* @__PURE__ */ new Set();
|
|
965
|
+
for (const format of value) {
|
|
966
|
+
if (!DATASET_FORMAT_SET.has(format)) throw new Error(`buildRlDataset: unsupported format ${JSON.stringify(format)}; expected exactly one of: grpo, sft, dpo`);
|
|
967
|
+
const datasetFormat = format;
|
|
968
|
+
if (seen.has(datasetFormat)) throw new Error(`buildRlDataset: duplicate format ${JSON.stringify(datasetFormat)}; each format may be requested once`);
|
|
969
|
+
seen.add(datasetFormat);
|
|
970
|
+
formats.push(datasetFormat);
|
|
971
|
+
}
|
|
972
|
+
return formats;
|
|
593
973
|
}
|
|
594
974
|
function distinct(xs) {
|
|
595
|
-
|
|
975
|
+
return [...new Set(xs.filter((x) => typeof x === "string" && x.length > 0))].sort();
|
|
596
976
|
}
|
|
597
977
|
function computeRewardStats(values) {
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
978
|
+
if (values.length === 0) return {
|
|
979
|
+
n: 0,
|
|
980
|
+
mean: null,
|
|
981
|
+
median: null,
|
|
982
|
+
min: null,
|
|
983
|
+
max: null,
|
|
984
|
+
std: null
|
|
985
|
+
};
|
|
986
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
987
|
+
const n = sorted.length;
|
|
988
|
+
const mean = sorted.reduce((s, x) => s + x, 0) / n;
|
|
989
|
+
const mid = Math.floor(n / 2);
|
|
990
|
+
const median = n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
|
|
991
|
+
const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
|
|
992
|
+
return {
|
|
993
|
+
n,
|
|
994
|
+
mean,
|
|
995
|
+
median,
|
|
996
|
+
min: sorted[0],
|
|
997
|
+
max: sorted[n - 1],
|
|
998
|
+
std: Math.sqrt(variance)
|
|
999
|
+
};
|
|
1000
|
+
}
|
|
1001
|
+
function computeStatsFromLines(lines) {
|
|
1002
|
+
const splits = {
|
|
1003
|
+
search: 0,
|
|
1004
|
+
dev: 0,
|
|
1005
|
+
holdout: 0,
|
|
1006
|
+
canary: 0
|
|
1007
|
+
};
|
|
1008
|
+
let inTok = 0;
|
|
1009
|
+
let outTok = 0;
|
|
1010
|
+
let cost = 0;
|
|
1011
|
+
let rolloutsWithoutCost = 0;
|
|
1012
|
+
const rewards = [];
|
|
1013
|
+
for (const line of lines) {
|
|
1014
|
+
splits[line.task.split] += 1;
|
|
1015
|
+
inTok += line.cost.tokens_in ?? 0;
|
|
1016
|
+
outTok += line.cost.tokens_out ?? 0;
|
|
1017
|
+
if (line.cost.usd === null) rolloutsWithoutCost++;
|
|
1018
|
+
else cost += line.cost.usd;
|
|
1019
|
+
const rw = trainableLineReward(line);
|
|
1020
|
+
if (rw !== null) rewards.push(rw);
|
|
1021
|
+
}
|
|
1022
|
+
return {
|
|
1023
|
+
records: lines.length,
|
|
1024
|
+
scoredRecords: rewards.length,
|
|
1025
|
+
splits,
|
|
1026
|
+
reward: computeRewardStats(rewards),
|
|
1027
|
+
models: distinct(lines.map((l) => l.policy.model)),
|
|
1028
|
+
promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),
|
|
1029
|
+
commitShas: distinct(lines.map((l) => l.policy.profile_commit)),
|
|
1030
|
+
totalTokens: {
|
|
1031
|
+
input: inTok,
|
|
1032
|
+
output: outTok
|
|
1033
|
+
},
|
|
1034
|
+
totalCostUsd: cost,
|
|
1035
|
+
rolloutsWithoutCost
|
|
1036
|
+
};
|
|
1037
|
+
}
|
|
1038
|
+
/**
|
|
1039
|
+
* Package graded rollout lines into a publishable RL dataset bundle: the
|
|
1040
|
+
* trainer-format JSONL files + a manifest + a datasheet. DPO requires
|
|
1041
|
+
* pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
|
|
1042
|
+
* the lines directly via the supplied lookups. Throws on an empty corpus —
|
|
1043
|
+
* an empty dataset must never be published.
|
|
1044
|
+
*/
|
|
1045
|
+
async function buildRlDataset(lines, lookups, config, preferences) {
|
|
1046
|
+
if (lines.length === 0) throw new Error("buildRlDataset: no rollout lines — refusing to package an empty dataset");
|
|
1047
|
+
const formats = validateDatasetFormats(config.formats === void 0 ? ["sft"] : config.formats);
|
|
1048
|
+
const files = {};
|
|
1049
|
+
const rowCounts = {};
|
|
1050
|
+
if (formats.includes("grpo")) {
|
|
1051
|
+
const rows = await toGrpoRows(lines, lookups);
|
|
1052
|
+
requireRows("grpo", rows.length);
|
|
1053
|
+
files["train.grpo.jsonl"] = toGrpoJsonl(rows);
|
|
1054
|
+
rowCounts.grpo = rows.length;
|
|
1055
|
+
}
|
|
1056
|
+
if (formats.includes("sft")) {
|
|
1057
|
+
const rows = await toSftRows(lines, lookups);
|
|
1058
|
+
requireRows("sft", rows.length);
|
|
1059
|
+
files["train.sft.jsonl"] = toSftJsonl(rows);
|
|
1060
|
+
rowCounts.sft = rows.length;
|
|
1061
|
+
}
|
|
1062
|
+
if (formats.includes("dpo")) {
|
|
1063
|
+
if (!preferences) throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
|
|
1064
|
+
const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines });
|
|
1065
|
+
requireRows("dpo", rows.length);
|
|
1066
|
+
files["train.dpo.jsonl"] = toDpoJsonl(rows);
|
|
1067
|
+
rowCounts.dpo = rows.length;
|
|
1068
|
+
}
|
|
1069
|
+
if (!Object.keys(files).some((name) => name.startsWith("train.") && name.endsWith(".jsonl"))) throw new Error("buildRlDataset: no trainer file was emitted");
|
|
1070
|
+
const manifest = {
|
|
1071
|
+
...config,
|
|
1072
|
+
formats,
|
|
1073
|
+
rowCounts,
|
|
1074
|
+
stats: computeStatsFromLines(lines)
|
|
1075
|
+
};
|
|
1076
|
+
files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}\n`;
|
|
1077
|
+
files["DATASHEET.md"] = datasheetToMarkdown(manifest);
|
|
1078
|
+
return {
|
|
1079
|
+
manifest,
|
|
1080
|
+
files
|
|
1081
|
+
};
|
|
676
1082
|
}
|
|
677
1083
|
function requireRows(format, rows) {
|
|
678
|
-
|
|
679
|
-
throw new Error(`buildRlDataset: requested '${format}' format produced no trainable rows`);
|
|
680
|
-
}
|
|
1084
|
+
if (rows === 0) throw new Error(`buildRlDataset: requested '${format}' format produced no trainable rows`);
|
|
681
1085
|
}
|
|
682
1086
|
function pct(x) {
|
|
683
|
-
|
|
1087
|
+
return `${(x * 100).toFixed(1)}%`;
|
|
684
1088
|
}
|
|
685
1089
|
function stat(value) {
|
|
686
|
-
|
|
1090
|
+
return value === null ? "n/a" : value.toFixed(3);
|
|
687
1091
|
}
|
|
1092
|
+
/** Render the "Datasheet for Datasets" card that a buyer reads. */
|
|
688
1093
|
function datasheetToMarkdown(m) {
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
1094
|
+
const s = m.stats;
|
|
1095
|
+
const total = s.records || 1;
|
|
1096
|
+
const splitLines = [
|
|
1097
|
+
"search",
|
|
1098
|
+
"dev",
|
|
1099
|
+
"holdout",
|
|
1100
|
+
"canary"
|
|
1101
|
+
].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
|
|
1102
|
+
const costNote = s.rolloutsWithoutCost > 0 ? ` (floor — ${s.rolloutsWithoutCost} rollout(s) never captured a cost)` : "";
|
|
1103
|
+
const deterministic = m.reward.kind === "deterministic";
|
|
1104
|
+
return [
|
|
1105
|
+
`# Dataset: ${m.name} \`v${m.version}\``,
|
|
1106
|
+
"",
|
|
1107
|
+
`**Domain:** ${m.domain} | **Created:** ${m.createdAtIso} | **License:** ${m.license}`,
|
|
1108
|
+
"",
|
|
1109
|
+
"## Reward provenance",
|
|
1110
|
+
`- **Kind:** ${m.reward.kind}${deterministic ? " (decidable, not judge noise)" : ""}`,
|
|
1111
|
+
`- **Source:** ${m.reward.source}`,
|
|
1112
|
+
`- **Description:** ${m.reward.description}`,
|
|
1113
|
+
"",
|
|
1114
|
+
"## Composition",
|
|
1115
|
+
`- **Records (trajectories):** ${s.records}`,
|
|
1116
|
+
`- **Scored records:** ${s.scoredRecords}`,
|
|
1117
|
+
`- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(", ")}`,
|
|
1118
|
+
"- **Splits:**",
|
|
1119
|
+
splitLines,
|
|
1120
|
+
"",
|
|
1121
|
+
"## Reward distribution",
|
|
1122
|
+
`- n=${s.reward.n} | mean=${stat(s.reward.mean)} | median=${stat(s.reward.median)} | min=${stat(s.reward.min)} | max=${stat(s.reward.max)} | std=${stat(s.reward.std)}`,
|
|
1123
|
+
"",
|
|
1124
|
+
"## Provenance",
|
|
1125
|
+
`- **Models:** ${s.models.join(", ")}`,
|
|
1126
|
+
`- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
|
|
1127
|
+
`- **Commits:** ${s.commitShas.join(", ")}`,
|
|
1128
|
+
`- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,
|
|
1129
|
+
"",
|
|
1130
|
+
"## Quality gates",
|
|
1131
|
+
`- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
|
|
1132
|
+
`- Dedup: ${m.qualityGates?.dedup ? "yes" : "no"} | Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? "yes" : "no"}`,
|
|
1133
|
+
"",
|
|
1134
|
+
"## Recommended uses",
|
|
1135
|
+
m.intendedUse,
|
|
1136
|
+
"",
|
|
1137
|
+
"## Out of scope",
|
|
1138
|
+
m.outOfScope,
|
|
1139
|
+
"",
|
|
1140
|
+
"## Limitations",
|
|
1141
|
+
m.limitations,
|
|
1142
|
+
"",
|
|
1143
|
+
"## Token rendering",
|
|
1144
|
+
"For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns. See `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.",
|
|
1145
|
+
""
|
|
1146
|
+
].join("\n");
|
|
1147
|
+
}
|
|
1148
|
+
//#endregion
|
|
1149
|
+
//#region src/rl/corpus.ts
|
|
1150
|
+
/**
|
|
1151
|
+
* RL corpus — the durable, append-only accumulation of graded RunRecords that
|
|
1152
|
+
* every eval run deposits BY DEFAULT.
|
|
1153
|
+
*
|
|
1154
|
+
* The dataset is the free exhaust of the normal eval process: we run evals
|
|
1155
|
+
* constantly to get an agent production-ready, and those runs already produce
|
|
1156
|
+
* graded trajectories. Instead of writing them to an ephemeral run dir and
|
|
1157
|
+
* throwing them away, `appendToCorpus` accumulates them into a durable corpus;
|
|
1158
|
+
* `buildDatasetFromCorpus` later harvests the whole corpus into a publishable
|
|
1159
|
+
* bundle. No separate data-collection campaign — the data accrues from work we
|
|
1160
|
+
* do anyway. This is the "best things for free by our process" layer.
|
|
1161
|
+
*
|
|
1162
|
+
* Trajectory text rides on the record as top-level `prompt` / `completion`
|
|
1163
|
+
* (what the eval harnesses capture; the RunRecord validator ignores the extra
|
|
1164
|
+
* keys). The harvest reads them directly — no trace store round-trip needed.
|
|
1165
|
+
*/
|
|
1166
|
+
/**
|
|
1167
|
+
* Append graded records to the corpus (append-only JSONL). Deduplicates by
|
|
1168
|
+
* `runId` against what's already on disk so re-running the same harness is
|
|
1169
|
+
* idempotent. Creates the file and parent dir. This is the call every eval
|
|
1170
|
+
* harness makes by default after producing its records.
|
|
1171
|
+
*/
|
|
739
1172
|
function appendToCorpus(records, corpusPath) {
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
1173
|
+
mkdirSync(dirname(corpusPath), { recursive: true });
|
|
1174
|
+
const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : [];
|
|
1175
|
+
const seen = new Set(existing.map((r) => r.runId));
|
|
1176
|
+
const lines = [];
|
|
1177
|
+
let appended = 0;
|
|
1178
|
+
let skipped = 0;
|
|
1179
|
+
for (const r of records) {
|
|
1180
|
+
if (seen.has(r.runId)) {
|
|
1181
|
+
skipped++;
|
|
1182
|
+
continue;
|
|
1183
|
+
}
|
|
1184
|
+
seen.add(r.runId);
|
|
1185
|
+
lines.push(JSON.stringify(r));
|
|
1186
|
+
appended++;
|
|
1187
|
+
}
|
|
1188
|
+
if (lines.length > 0) appendFileSync(corpusPath, `${lines.join("\n")}\n`);
|
|
1189
|
+
return {
|
|
1190
|
+
appended,
|
|
1191
|
+
skipped,
|
|
1192
|
+
total: existing.length + appended
|
|
1193
|
+
};
|
|
1194
|
+
}
|
|
1195
|
+
/** Read the full corpus. Returns [] if the corpus does not exist yet. */
|
|
759
1196
|
function readCorpus(corpusPath) {
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
1197
|
+
if (!existsSync(corpusPath)) return [];
|
|
1198
|
+
const out = [];
|
|
1199
|
+
for (const line of readFileSync(corpusPath, "utf8").split("\n")) if (line.trim()) out.push(JSON.parse(line));
|
|
1200
|
+
return out;
|
|
1201
|
+
}
|
|
1202
|
+
/**
|
|
1203
|
+
* The harvest's score reader is GATED: a gamed run reads 0, so it cannot buy
|
|
1204
|
+
* its way past `minScore` into the published bundle with its claimed score.
|
|
1205
|
+
* `null` = unscored (a labeled gap, dropped before packaging, never a 0).
|
|
1206
|
+
*/
|
|
767
1207
|
function rewardOf(r) {
|
|
768
|
-
|
|
769
|
-
|
|
1208
|
+
const v = trainingScore(r);
|
|
1209
|
+
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
1210
|
+
}
|
|
1211
|
+
/**
|
|
1212
|
+
* Harvest the accumulated corpus into a publishable RL dataset bundle. Reads
|
|
1213
|
+
* trajectory text from each record's top-level `prompt`/`completion`; records
|
|
1214
|
+
* missing either are excluded (a graded score with no trajectory can't train).
|
|
1215
|
+
* Optionally filters by score / split. Throws (via buildRlDataset) if nothing
|
|
1216
|
+
* survives — an empty dataset must never be published.
|
|
1217
|
+
*
|
|
1218
|
+
* `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run
|
|
1219
|
+
* cannot buy its way into the published bundle with its claimed score —
|
|
1220
|
+
* `minScore` is exactly the door a reward-hacked run would otherwise clear for
|
|
1221
|
+
* SFT. Unscored records are dropped before packaging: a missing label is not a
|
|
1222
|
+
* zero, and it is not publishable either.
|
|
1223
|
+
*/
|
|
770
1224
|
async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
1225
|
+
let records = readCorpus(corpusPath).filter((r) => typeof r.prompt === "string" && typeof r.completion === "string");
|
|
1226
|
+
if (opts.splits) records = records.filter((r) => opts.splits.includes(r.splitTag));
|
|
1227
|
+
records = records.filter((r) => rewardOf(r) !== null);
|
|
1228
|
+
if (opts.minScore != null) records = records.filter((r) => {
|
|
1229
|
+
const reward = rewardOf(r);
|
|
1230
|
+
return reward !== null && reward >= opts.minScore;
|
|
1231
|
+
});
|
|
1232
|
+
const text = new Map(records.map((r) => [r.runId, {
|
|
1233
|
+
prompt: r.prompt,
|
|
1234
|
+
completion: r.completion
|
|
1235
|
+
}]));
|
|
1236
|
+
const lookups = {
|
|
1237
|
+
promptOf: (id) => text.get(id)?.prompt ?? "",
|
|
1238
|
+
completionOf: (id) => text.get(id)?.completion ?? "",
|
|
1239
|
+
allowHeldOutTrainingData: opts.allowHeldOutTrainingData
|
|
1240
|
+
};
|
|
1241
|
+
const { rows } = await mintRolloutRows(records, new InMemoryTraceStore());
|
|
1242
|
+
return buildRlDataset(rows, lookups, config);
|
|
1243
|
+
}
|
|
1244
|
+
//#endregion
|
|
1245
|
+
//#region src/rl/predictive-validity-researcher.ts
|
|
1246
|
+
/**
|
|
1247
|
+
* Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
|
|
1248
|
+
* rubrics that don't predict deployment outcomes don't earn weight.
|
|
1249
|
+
*/
|
|
794
1250
|
var PredictiveValidityResearcher = class {
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
setReport(report) {
|
|
933
|
-
this.lastReport = report;
|
|
934
|
-
}
|
|
935
|
-
getLastReport() {
|
|
936
|
-
return this.lastReport;
|
|
937
|
-
}
|
|
1251
|
+
opts;
|
|
1252
|
+
lastReport = null;
|
|
1253
|
+
constructor(opts) {
|
|
1254
|
+
this.opts = opts;
|
|
1255
|
+
}
|
|
1256
|
+
async inspectFailures(runs) {
|
|
1257
|
+
const threshold = this.opts.failureThreshold ?? .5;
|
|
1258
|
+
const failures = [];
|
|
1259
|
+
const failingRuns = runs.filter((r) => {
|
|
1260
|
+
const score = runTaskScore(r);
|
|
1261
|
+
return typeof score === "number" && score < threshold;
|
|
1262
|
+
});
|
|
1263
|
+
if (failingRuns.length === 0) return failures;
|
|
1264
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
1265
|
+
for (const r of failingRuns) {
|
|
1266
|
+
const arr = grouped.get(r.candidateId) ?? [];
|
|
1267
|
+
arr.push(r);
|
|
1268
|
+
grouped.set(r.candidateId, arr);
|
|
1269
|
+
}
|
|
1270
|
+
for (const [candidateId, group] of grouped.entries()) {
|
|
1271
|
+
const meanScore = group.reduce((s, r) => {
|
|
1272
|
+
const score = runTaskScore(r);
|
|
1273
|
+
if (score === void 0) throw new Error(`failing run ${r.runId} unexpectedly has no task score`);
|
|
1274
|
+
return s + score;
|
|
1275
|
+
}, 0) / group.length;
|
|
1276
|
+
failures.push({
|
|
1277
|
+
code: `low-score-${candidateId}`,
|
|
1278
|
+
description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,
|
|
1279
|
+
evidence: {
|
|
1280
|
+
runIds: group.slice(0, 8).map((r) => r.runId),
|
|
1281
|
+
samples: group.length
|
|
1282
|
+
}
|
|
1283
|
+
});
|
|
1284
|
+
}
|
|
1285
|
+
return failures;
|
|
1286
|
+
}
|
|
1287
|
+
async proposeChange(failures) {
|
|
1288
|
+
if (failures.length === 0) return [];
|
|
1289
|
+
if (this.lastReport === null) return [{
|
|
1290
|
+
kind: "threshold",
|
|
1291
|
+
payload: { directive: "researcher.collect-more-outcomes" },
|
|
1292
|
+
rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
|
|
1293
|
+
}];
|
|
1294
|
+
const decorativeThreshold = this.opts.decorativeThreshold ?? .4;
|
|
1295
|
+
const changes = [];
|
|
1296
|
+
for (const ranking of this.lastReport.ranked) {
|
|
1297
|
+
if (ranking.verdict === "load_bearing") continue;
|
|
1298
|
+
if (Math.abs(ranking.spearman) >= decorativeThreshold) continue;
|
|
1299
|
+
changes.push({
|
|
1300
|
+
kind: "reviewer_prompt",
|
|
1301
|
+
payload: {
|
|
1302
|
+
rubric: ranking.rubric,
|
|
1303
|
+
action: "down-weight",
|
|
1304
|
+
spearman: ranking.spearman,
|
|
1305
|
+
bestOutcome: ranking.bestOutcome
|
|
1306
|
+
},
|
|
1307
|
+
rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,
|
|
1308
|
+
expectedDelta: -Math.max(0, .05 - Math.abs(ranking.spearman))
|
|
1309
|
+
});
|
|
1310
|
+
}
|
|
1311
|
+
for (const ranking of this.lastReport.ranked.slice(0, 1)) {
|
|
1312
|
+
if (ranking.verdict !== "load_bearing") continue;
|
|
1313
|
+
changes.push({
|
|
1314
|
+
kind: "reviewer_prompt",
|
|
1315
|
+
payload: {
|
|
1316
|
+
rubric: ranking.rubric,
|
|
1317
|
+
action: "up-weight",
|
|
1318
|
+
spearman: ranking.spearman,
|
|
1319
|
+
bestOutcome: ranking.bestOutcome
|
|
1320
|
+
},
|
|
1321
|
+
rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,
|
|
1322
|
+
expectedDelta: Math.max(0, Math.abs(ranking.spearman) - .5) * .1
|
|
1323
|
+
});
|
|
1324
|
+
}
|
|
1325
|
+
return changes;
|
|
1326
|
+
}
|
|
1327
|
+
async applyChange(changes, baseline) {
|
|
1328
|
+
return {
|
|
1329
|
+
...baseline,
|
|
1330
|
+
changes: [...baseline.changes, ...changes]
|
|
1331
|
+
};
|
|
1332
|
+
}
|
|
1333
|
+
async evaluateChange(plan) {
|
|
1334
|
+
return {
|
|
1335
|
+
plan,
|
|
1336
|
+
runs: [],
|
|
1337
|
+
gateDecision: {
|
|
1338
|
+
promote: false,
|
|
1339
|
+
candidateId: plan.proposedCandidateId,
|
|
1340
|
+
baselineId: plan.baselineCandidateId,
|
|
1341
|
+
evidence: {
|
|
1342
|
+
productiveRuns: 0,
|
|
1343
|
+
unpairedCandidateRuns: 0,
|
|
1344
|
+
unpairedBaselineRuns: 0,
|
|
1345
|
+
medianPairedDelta: null,
|
|
1346
|
+
pairedCI: null,
|
|
1347
|
+
pairedPValue: null,
|
|
1348
|
+
searchScore: null,
|
|
1349
|
+
holdoutScore: null,
|
|
1350
|
+
overfitGap: null,
|
|
1351
|
+
baselineOverfitGap: null,
|
|
1352
|
+
medianCandidateCost: null,
|
|
1353
|
+
medianBaselineCost: null,
|
|
1354
|
+
realnessGatedRuns: 0
|
|
1355
|
+
},
|
|
1356
|
+
reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
|
|
1357
|
+
rejectionCode: "few_runs"
|
|
1358
|
+
}
|
|
1359
|
+
};
|
|
1360
|
+
}
|
|
1361
|
+
/**
|
|
1362
|
+
* Run the predictive-validity check explicitly against a fresh RunRecord
|
|
1363
|
+
* set. Updates the researcher's cached report so subsequent
|
|
1364
|
+
* `proposeChange` calls have evidence to draw from.
|
|
1365
|
+
*/
|
|
1366
|
+
async runValidityCheck(runs) {
|
|
1367
|
+
const report = await rubricPredictiveValidity({
|
|
1368
|
+
runs,
|
|
1369
|
+
outcomes: this.opts.outcomes,
|
|
1370
|
+
outcomeMetrics: this.opts.outcomeMetrics,
|
|
1371
|
+
rubrics: this.opts.rubrics
|
|
1372
|
+
});
|
|
1373
|
+
if (this.opts.onReport) await this.opts.onReport(report);
|
|
1374
|
+
this.lastReport = report;
|
|
1375
|
+
return report;
|
|
1376
|
+
}
|
|
1377
|
+
/**
|
|
1378
|
+
* Force-feed a predictive-validity report into the researcher state —
|
|
1379
|
+
* useful when the consumer ran the report out-of-band and wants the
|
|
1380
|
+
* researcher's later proposals informed by it.
|
|
1381
|
+
*/
|
|
1382
|
+
setReport(report) {
|
|
1383
|
+
this.lastReport = report;
|
|
1384
|
+
}
|
|
1385
|
+
getLastReport() {
|
|
1386
|
+
return this.lastReport;
|
|
1387
|
+
}
|
|
938
1388
|
};
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
1389
|
+
//#endregion
|
|
1390
|
+
//#region src/rl/preferences.ts
|
|
1391
|
+
/** The split each path pairs by default: training data comes from search. */
|
|
1392
|
+
const SPLIT_DEFAULT = "search";
|
|
1393
|
+
/**
|
|
1394
|
+
* Convert rollout lines to preference triples for RL training.
|
|
1395
|
+
*
|
|
1396
|
+
* Returns a structured report so callers can see how much data was
|
|
1397
|
+
* dropped and why (low-margin pairs, singleton cells). For production
|
|
1398
|
+
* pipelines, you usually want to:
|
|
1399
|
+
*
|
|
1400
|
+
* 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
|
|
1401
|
+
* 2. Mint the runs with `mintRolloutRows` and call this with
|
|
1402
|
+
* `strategy: 'paired-by-scenario-and-seed'`
|
|
1403
|
+
* 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with
|
|
1404
|
+
* prompt/completion resolvers and pipe to your DPO trainer
|
|
1405
|
+
*
|
|
1406
|
+
* The gate is what makes a preference dataset safe: ordered on an ungated
|
|
1407
|
+
* score, a gamed run with an inflated number becomes the `chosen` side and DPO
|
|
1408
|
+
* is trained to prefer the gaming trajectory over its honest sibling. A gated
|
|
1409
|
+
* line arrives here already scored 0, so it sinks to `rejected`.
|
|
1410
|
+
*/
|
|
1411
|
+
function extractPreferences(lines, opts = {}) {
|
|
1412
|
+
const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
|
|
1413
|
+
const minMargin = opts.minMargin ?? .05;
|
|
1414
|
+
const requestedSplit = opts.split;
|
|
1415
|
+
if (requestedSplit === "holdout" && opts.allowHeldOutTrainingData !== true) throw new Error("extractPreferences: split \"holdout\" requires allowHeldOutTrainingData: true");
|
|
1416
|
+
if (requestedSplit === "dev" || requestedSplit === "canary") throw new Error(`extractPreferences: split "${requestedSplit}" is evaluation-only; train from "search"`);
|
|
1417
|
+
const candidates = candidatesFromLines(lines, opts);
|
|
1418
|
+
return {
|
|
1419
|
+
...pairCandidates(candidates.rows, strategy, minMargin),
|
|
1420
|
+
linesWithoutCandidateId: candidates.withoutCandidateId
|
|
1421
|
+
};
|
|
1422
|
+
}
|
|
1423
|
+
function candidatesFromLines(lines, opts) {
|
|
1424
|
+
const split = opts.split ?? SPLIT_DEFAULT;
|
|
1425
|
+
const rows = [];
|
|
1426
|
+
let withoutCandidateId = 0;
|
|
1427
|
+
for (const line of lines) {
|
|
1428
|
+
if (line.task.split !== split) continue;
|
|
1429
|
+
if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) continue;
|
|
1430
|
+
const score = trainableLineReward(line);
|
|
1431
|
+
if (score === null) continue;
|
|
1432
|
+
const candidateId = line.candidate_id;
|
|
1433
|
+
if (candidateId === null || candidateId === void 0 || candidateId.length === 0) {
|
|
1434
|
+
withoutCandidateId++;
|
|
1435
|
+
continue;
|
|
1436
|
+
}
|
|
1437
|
+
rows.push({
|
|
1438
|
+
scenarioId: line.task.instance_id,
|
|
1439
|
+
runId: line.run_id,
|
|
1440
|
+
candidateId,
|
|
1441
|
+
seed: line.task.seed,
|
|
1442
|
+
score,
|
|
1443
|
+
promptHash: line.policy.prompt_hash ?? "",
|
|
1444
|
+
configHash: line.policy.config_hash ?? "",
|
|
1445
|
+
model: line.policy.model ?? ""
|
|
1446
|
+
});
|
|
1447
|
+
}
|
|
1448
|
+
return {
|
|
1449
|
+
rows,
|
|
1450
|
+
withoutCandidateId
|
|
1451
|
+
};
|
|
1452
|
+
}
|
|
1453
|
+
function pairCandidates(scoredEntries, strategy, minMargin) {
|
|
1454
|
+
const pairs = [];
|
|
1455
|
+
let pairsBelowMargin = 0;
|
|
1456
|
+
let cellsSingleton = 0;
|
|
1457
|
+
let cellsInspected = 0;
|
|
1458
|
+
if (strategy === "paired-by-scenario-and-seed") {
|
|
1459
|
+
const groups = /* @__PURE__ */ new Map();
|
|
1460
|
+
for (const e of scoredEntries) {
|
|
1461
|
+
const key = `${e.scenarioId}::${e.seed}`;
|
|
1462
|
+
const arr = groups.get(key) ?? [];
|
|
1463
|
+
arr.push(e);
|
|
1464
|
+
groups.set(key, arr);
|
|
1465
|
+
}
|
|
1466
|
+
for (const members of groups.values()) {
|
|
1467
|
+
cellsInspected++;
|
|
1468
|
+
if (members.length < 2) {
|
|
1469
|
+
cellsSingleton++;
|
|
1470
|
+
continue;
|
|
1471
|
+
}
|
|
1472
|
+
for (let i = 0; i < members.length; i++) for (let j = i + 1; j < members.length; j++) {
|
|
1473
|
+
const a = members[i];
|
|
1474
|
+
const b = members[j];
|
|
1475
|
+
if (a.candidateId === b.candidateId) continue;
|
|
1476
|
+
const result = makePair(a, b, a.scenarioId, minMargin);
|
|
1477
|
+
if (result.kind === "admit") pairs.push(result.pair);
|
|
1478
|
+
else pairsBelowMargin++;
|
|
1479
|
+
}
|
|
1480
|
+
}
|
|
1481
|
+
} else if (strategy === "paired-by-scenario") {
|
|
1482
|
+
const byScenarioVariant = /* @__PURE__ */ new Map();
|
|
1483
|
+
for (const e of scoredEntries) {
|
|
1484
|
+
let perScenario = byScenarioVariant.get(e.scenarioId);
|
|
1485
|
+
if (!perScenario) {
|
|
1486
|
+
perScenario = /* @__PURE__ */ new Map();
|
|
1487
|
+
byScenarioVariant.set(e.scenarioId, perScenario);
|
|
1488
|
+
}
|
|
1489
|
+
const cur = perScenario.get(e.candidateId);
|
|
1490
|
+
if (cur) {
|
|
1491
|
+
cur.sum += e.score;
|
|
1492
|
+
cur.n++;
|
|
1493
|
+
} else perScenario.set(e.candidateId, {
|
|
1494
|
+
entry: e,
|
|
1495
|
+
sum: e.score,
|
|
1496
|
+
n: 1
|
|
1497
|
+
});
|
|
1498
|
+
}
|
|
1499
|
+
for (const [sid, perVariant] of byScenarioVariant.entries()) {
|
|
1500
|
+
cellsInspected++;
|
|
1501
|
+
const arr = [...perVariant.values()].map((agg) => ({
|
|
1502
|
+
...agg.entry,
|
|
1503
|
+
score: agg.sum / agg.n
|
|
1504
|
+
}));
|
|
1505
|
+
if (arr.length < 2) {
|
|
1506
|
+
cellsSingleton++;
|
|
1507
|
+
continue;
|
|
1508
|
+
}
|
|
1509
|
+
for (let i = 0; i < arr.length; i++) for (let j = i + 1; j < arr.length; j++) {
|
|
1510
|
+
const result = makePair(arr[i], arr[j], sid, minMargin);
|
|
1511
|
+
if (result.kind === "admit") pairs.push(result.pair);
|
|
1512
|
+
else pairsBelowMargin++;
|
|
1513
|
+
}
|
|
1514
|
+
}
|
|
1515
|
+
} else {
|
|
1516
|
+
const byScenario = /* @__PURE__ */ new Map();
|
|
1517
|
+
for (const e of scoredEntries) {
|
|
1518
|
+
const arr = byScenario.get(e.scenarioId) ?? [];
|
|
1519
|
+
arr.push(e);
|
|
1520
|
+
byScenario.set(e.scenarioId, arr);
|
|
1521
|
+
}
|
|
1522
|
+
for (const [sid, arr] of byScenario.entries()) {
|
|
1523
|
+
cellsInspected++;
|
|
1524
|
+
if (arr.length < 2) {
|
|
1525
|
+
cellsSingleton++;
|
|
1526
|
+
continue;
|
|
1527
|
+
}
|
|
1528
|
+
const sorted = [...arr].sort((a, b) => a.score - b.score);
|
|
1529
|
+
const top = sorted[sorted.length - 1];
|
|
1530
|
+
const bot = sorted[0];
|
|
1531
|
+
if (top.candidateId === bot.candidateId) {
|
|
1532
|
+
cellsSingleton++;
|
|
1533
|
+
continue;
|
|
1534
|
+
}
|
|
1535
|
+
const result = makePair(bot, top, sid, minMargin);
|
|
1536
|
+
if (result.kind === "admit") pairs.push(result.pair);
|
|
1537
|
+
else pairsBelowMargin++;
|
|
1538
|
+
}
|
|
1539
|
+
}
|
|
1540
|
+
return {
|
|
1541
|
+
pairs,
|
|
1542
|
+
cellsInspected,
|
|
1543
|
+
pairsBelowMargin,
|
|
1544
|
+
cellsSingleton,
|
|
1545
|
+
strategy
|
|
1546
|
+
};
|
|
1547
|
+
}
|
|
1548
|
+
const PREFERENCE_RUN_IDS = (t) => [t.chosenRunId, t.rejectedRunId];
|
|
1549
|
+
const TRL_CONTEXT_REQUIREMENT = {
|
|
1550
|
+
exporter: "TRL preference export",
|
|
1551
|
+
contextType: "RolloutLineContext",
|
|
1552
|
+
because: "a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair."
|
|
1553
|
+
};
|
|
1554
|
+
const ANTHROPIC_CONTEXT_REQUIREMENT = {
|
|
1555
|
+
exporter: "Anthropic preference export",
|
|
1556
|
+
contextType: "RolloutLineContext",
|
|
1557
|
+
because: "a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one."
|
|
943
1558
|
};
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
for (const e of scoredEntries) {
|
|
997
|
-
const sid = scenarioOf(e.run);
|
|
998
|
-
let perScenario = byScenarioVariant.get(sid);
|
|
999
|
-
if (!perScenario) {
|
|
1000
|
-
perScenario = /* @__PURE__ */ new Map();
|
|
1001
|
-
byScenarioVariant.set(sid, perScenario);
|
|
1002
|
-
}
|
|
1003
|
-
const cur = perScenario.get(e.run.candidateId);
|
|
1004
|
-
if (cur) {
|
|
1005
|
-
cur.sum += e.score;
|
|
1006
|
-
cur.n++;
|
|
1007
|
-
} else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
|
|
1008
|
-
}
|
|
1009
|
-
for (const [sid, perVariant] of byScenarioVariant.entries()) {
|
|
1010
|
-
cellsInspected++;
|
|
1011
|
-
const arr = [...perVariant.entries()].map(([vid, agg]) => ({
|
|
1012
|
-
run: agg.run,
|
|
1013
|
-
score: agg.sum / agg.n,
|
|
1014
|
-
variantId: vid
|
|
1015
|
-
}));
|
|
1016
|
-
if (arr.length < 2) {
|
|
1017
|
-
cellsSingleton++;
|
|
1018
|
-
continue;
|
|
1019
|
-
}
|
|
1020
|
-
for (let i = 0; i < arr.length; i++) {
|
|
1021
|
-
for (let j = i + 1; j < arr.length; j++) {
|
|
1022
|
-
const result = makePair(arr[i], arr[j], sid, minMargin);
|
|
1023
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
1024
|
-
else pairsBelowMargin++;
|
|
1025
|
-
}
|
|
1026
|
-
}
|
|
1027
|
-
}
|
|
1028
|
-
} else {
|
|
1029
|
-
const byScenario = /* @__PURE__ */ new Map();
|
|
1030
|
-
for (const e of scoredEntries) {
|
|
1031
|
-
const sid = scenarioOf(e.run);
|
|
1032
|
-
const arr = byScenario.get(sid) ?? [];
|
|
1033
|
-
arr.push(e);
|
|
1034
|
-
byScenario.set(sid, arr);
|
|
1035
|
-
}
|
|
1036
|
-
for (const [sid, arr] of byScenario.entries()) {
|
|
1037
|
-
cellsInspected++;
|
|
1038
|
-
if (arr.length < 2) {
|
|
1039
|
-
cellsSingleton++;
|
|
1040
|
-
continue;
|
|
1041
|
-
}
|
|
1042
|
-
const sorted = [...arr].sort((a, b) => a.score - b.score);
|
|
1043
|
-
const top = sorted[sorted.length - 1];
|
|
1044
|
-
const bot = sorted[0];
|
|
1045
|
-
if (top.run.candidateId === bot.run.candidateId) {
|
|
1046
|
-
cellsSingleton++;
|
|
1047
|
-
continue;
|
|
1048
|
-
}
|
|
1049
|
-
const result = makePair(bot, top, sid, minMargin);
|
|
1050
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
1051
|
-
else pairsBelowMargin++;
|
|
1052
|
-
}
|
|
1053
|
-
}
|
|
1054
|
-
return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
|
|
1055
|
-
}
|
|
1056
|
-
function toAnthropicFormat(triples) {
|
|
1057
|
-
return triples.map((t) => ({
|
|
1058
|
-
scenarioId: t.scenarioId,
|
|
1059
|
-
chosenRunId: t.chosenRunId,
|
|
1060
|
-
rejectedRunId: t.rejectedRunId,
|
|
1061
|
-
margin: t.marginScore
|
|
1062
|
-
}));
|
|
1559
|
+
/**
|
|
1560
|
+
* TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
|
|
1561
|
+
* where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes
|
|
1562
|
+
* would optimize the policy toward emitting hex digests. Neither the prompt
|
|
1563
|
+
* nor the completions live on the triple (it carries only run ids and hashes),
|
|
1564
|
+
* so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`
|
|
1565
|
+
* takes, keyed by run id, and this function resolves real text.
|
|
1566
|
+
*
|
|
1567
|
+
* The chosen and rejected sides of a valid pair share one prompt; resolving
|
|
1568
|
+
* both and comparing catches lookup bugs (a stale map keyed by the wrong id)
|
|
1569
|
+
* before they ship a row whose prompt does not match its rejected completion.
|
|
1570
|
+
*
|
|
1571
|
+
* `context` is REQUIRED: this is the third exporter over the identical
|
|
1572
|
+
* line-less input class, and the round that hardened `toPrmRows` while leaving
|
|
1573
|
+
* `toDpoRows` open is why every one of them now takes the same argument and
|
|
1574
|
+
* runs the same admission rule.
|
|
1575
|
+
*/
|
|
1576
|
+
async function toTRLFormat(triples, lookups, context) {
|
|
1577
|
+
const admitted = admitUngatedByInvocation(triples, PREFERENCE_RUN_IDS, context, TRL_CONTEXT_REQUIREMENT);
|
|
1578
|
+
const out = [];
|
|
1579
|
+
for (const t of admitted) {
|
|
1580
|
+
const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
|
|
1581
|
+
Promise.resolve(lookups.promptOf(t.chosenRunId)),
|
|
1582
|
+
Promise.resolve(lookups.promptOf(t.rejectedRunId)),
|
|
1583
|
+
Promise.resolve(lookups.completionOf(t.chosenRunId)),
|
|
1584
|
+
Promise.resolve(lookups.completionOf(t.rejectedRunId))
|
|
1585
|
+
]);
|
|
1586
|
+
if (chosenPrompt !== rejectedPrompt) throw new Error(`toTRLFormat: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`);
|
|
1587
|
+
out.push({
|
|
1588
|
+
prompt: chosenPrompt,
|
|
1589
|
+
chosen,
|
|
1590
|
+
rejected
|
|
1591
|
+
});
|
|
1592
|
+
}
|
|
1593
|
+
return out;
|
|
1594
|
+
}
|
|
1595
|
+
/**
|
|
1596
|
+
* Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
|
|
1597
|
+
* shape. Same caveat as TRL: prompt + outputs are content the caller has
|
|
1598
|
+
* to map back from the run record / raw event log.
|
|
1599
|
+
*
|
|
1600
|
+
* `context` is REQUIRED — see `toTRLFormat`. The emitted `margin` is a number
|
|
1601
|
+
* derived from the two runs' rewards, so this row is training signal even
|
|
1602
|
+
* though it ships no completion text.
|
|
1603
|
+
*/
|
|
1604
|
+
function toAnthropicFormat(triples, context) {
|
|
1605
|
+
return admitUngatedByInvocation(triples, PREFERENCE_RUN_IDS, context, ANTHROPIC_CONTEXT_REQUIREMENT).map((t) => ({
|
|
1606
|
+
scenarioId: t.scenarioId,
|
|
1607
|
+
chosenRunId: t.chosenRunId,
|
|
1608
|
+
rejectedRunId: t.rejectedRunId,
|
|
1609
|
+
margin: t.marginScore
|
|
1610
|
+
}));
|
|
1063
1611
|
}
|
|
1064
1612
|
function makePair(a, b, scenarioId, minMargin) {
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
}
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
}
|
|
1093
|
-
|
|
1094
|
-
|
|
1613
|
+
if (Math.abs(a.score - b.score) < minMargin) return { kind: "reject" };
|
|
1614
|
+
const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
|
|
1615
|
+
const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : void 0;
|
|
1616
|
+
return {
|
|
1617
|
+
kind: "admit",
|
|
1618
|
+
pair: {
|
|
1619
|
+
scenarioId,
|
|
1620
|
+
chosenRunId: chosen.runId,
|
|
1621
|
+
rejectedRunId: rejected.runId,
|
|
1622
|
+
chosenVariantId: chosen.candidateId,
|
|
1623
|
+
rejectedVariantId: rejected.candidateId,
|
|
1624
|
+
marginScore: chosen.score - rejected.score,
|
|
1625
|
+
scores: {
|
|
1626
|
+
chosen: chosen.score,
|
|
1627
|
+
rejected: rejected.score
|
|
1628
|
+
},
|
|
1629
|
+
seed,
|
|
1630
|
+
meta: {
|
|
1631
|
+
chosenPromptHash: chosen.promptHash,
|
|
1632
|
+
rejectedPromptHash: rejected.promptHash,
|
|
1633
|
+
chosenConfigHash: chosen.configHash,
|
|
1634
|
+
rejectedConfigHash: rejected.configHash,
|
|
1635
|
+
chosenModel: chosen.model,
|
|
1636
|
+
rejectedModel: rejected.model
|
|
1637
|
+
}
|
|
1638
|
+
}
|
|
1639
|
+
};
|
|
1640
|
+
}
|
|
1641
|
+
//#endregion
|
|
1642
|
+
//#region src/rl/process-reward.ts
|
|
1095
1643
|
async function extractStepRewards(store, runId, opts) {
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
return out;
|
|
1644
|
+
const ordered = [...await store.spans({ runId })].sort((a, b) => a.startedAt - b.startedAt);
|
|
1645
|
+
const out = [];
|
|
1646
|
+
let idx = 0;
|
|
1647
|
+
for (const span of ordered) {
|
|
1648
|
+
if (opts.preFilter && !opts.preFilter(span)) continue;
|
|
1649
|
+
let scored = null;
|
|
1650
|
+
for (const s of opts.scorers) {
|
|
1651
|
+
if (!s.appliesTo.includes(span.kind)) continue;
|
|
1652
|
+
const r = await s.score(span);
|
|
1653
|
+
if (r) {
|
|
1654
|
+
scored = r;
|
|
1655
|
+
break;
|
|
1656
|
+
}
|
|
1657
|
+
}
|
|
1658
|
+
if (!scored) continue;
|
|
1659
|
+
out.push({
|
|
1660
|
+
spanId: span.spanId,
|
|
1661
|
+
runId,
|
|
1662
|
+
stepIndex: idx++,
|
|
1663
|
+
kind: span.kind,
|
|
1664
|
+
name: span.name,
|
|
1665
|
+
reward: scored.reward,
|
|
1666
|
+
determinism: scored.determinism,
|
|
1667
|
+
rationale: scored.rationale,
|
|
1668
|
+
weight: scored.weight
|
|
1669
|
+
});
|
|
1670
|
+
}
|
|
1671
|
+
return out;
|
|
1125
1672
|
}
|
|
1126
1673
|
function runwiseStepRewardSummary(stepRewards) {
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1674
|
+
if (stepRewards.length === 0) return {
|
|
1675
|
+
runId: "",
|
|
1676
|
+
totalSteps: 0,
|
|
1677
|
+
meanReward: 0,
|
|
1678
|
+
sumWeightedReward: 0,
|
|
1679
|
+
failureFraction: 0,
|
|
1680
|
+
worstStepDelta: 0,
|
|
1681
|
+
worstStepIndex: null
|
|
1682
|
+
};
|
|
1683
|
+
const runId = stepRewards[0].runId;
|
|
1684
|
+
let sumW = 0;
|
|
1685
|
+
let sumWR = 0;
|
|
1686
|
+
let failures = 0;
|
|
1687
|
+
let worstDelta = 0;
|
|
1688
|
+
let worstIdx = null;
|
|
1689
|
+
let prev = stepRewards[0].reward;
|
|
1690
|
+
for (let i = 0; i < stepRewards.length; i++) {
|
|
1691
|
+
const s = stepRewards[i];
|
|
1692
|
+
const w = s.weight ?? 1;
|
|
1693
|
+
sumW += w;
|
|
1694
|
+
sumWR += w * s.reward;
|
|
1695
|
+
if (s.reward < .5) failures++;
|
|
1696
|
+
if (i > 0) {
|
|
1697
|
+
const delta = s.reward - prev;
|
|
1698
|
+
if (delta < worstDelta) {
|
|
1699
|
+
worstDelta = delta;
|
|
1700
|
+
worstIdx = i;
|
|
1701
|
+
}
|
|
1702
|
+
prev = s.reward;
|
|
1703
|
+
} else prev = s.reward;
|
|
1704
|
+
}
|
|
1705
|
+
return {
|
|
1706
|
+
runId,
|
|
1707
|
+
totalSteps: stepRewards.length,
|
|
1708
|
+
meanReward: sumW === 0 ? 0 : sumWR / sumW,
|
|
1709
|
+
sumWeightedReward: sumWR,
|
|
1710
|
+
failureFraction: failures / stepRewards.length,
|
|
1711
|
+
worstStepDelta: worstDelta,
|
|
1712
|
+
worstStepIndex: worstIdx
|
|
1713
|
+
};
|
|
1714
|
+
}
|
|
1715
|
+
/**
|
|
1716
|
+
* Build PRM training triples. The shape: pair runs that share an early
|
|
1717
|
+
* prefix (same scenario, same first N steps) and diverge later — at the
|
|
1718
|
+
* point of divergence, the high-reward run's next step is `chosen`, the
|
|
1719
|
+
* low-reward run's next step is `rejected`. This is the canonical PRM
|
|
1720
|
+
* training data shape from Lightman et al. and DeepSeek-R1 process
|
|
1721
|
+
* supervision.
|
|
1722
|
+
*
|
|
1723
|
+
* Implementation note: we don't have a way to detect "same prefix" in
|
|
1724
|
+
* the general agent setting (token-level prefixes require hashing model
|
|
1725
|
+
* outputs). The current heuristic groups by `(scenarioId, prefixSpanName
|
|
1726
|
+
* sequence)` — runs are paired when their first K span names match. For
|
|
1727
|
+
* production use this should be replaced with a proper trajectory-prefix
|
|
1728
|
+
* hash; the heuristic is good enough for early-stage scaffolding.
|
|
1729
|
+
*/
|
|
1172
1730
|
function prmTrainingPairs(stepRewardsByRun, opts = {}) {
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
}
|
|
1218
|
-
|
|
1219
|
-
|
|
1731
|
+
const minMargin = opts.minMargin ?? .2;
|
|
1732
|
+
const minPrefix = opts.minPrefixLength ?? 1;
|
|
1733
|
+
const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({
|
|
1734
|
+
runId,
|
|
1735
|
+
steps
|
|
1736
|
+
}));
|
|
1737
|
+
const triples = [];
|
|
1738
|
+
for (let i = 0; i < runs.length; i++) for (let j = i + 1; j < runs.length; j++) {
|
|
1739
|
+
const a = runs[i];
|
|
1740
|
+
const b = runs[j];
|
|
1741
|
+
const minLen = Math.min(a.steps.length, b.steps.length);
|
|
1742
|
+
if (minLen < minPrefix + 1) continue;
|
|
1743
|
+
let divergenceIdx = -1;
|
|
1744
|
+
for (let k = 0; k < minLen; k++) {
|
|
1745
|
+
const sa = a.steps[k];
|
|
1746
|
+
const sb = b.steps[k];
|
|
1747
|
+
const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name;
|
|
1748
|
+
const rewardGap = Math.abs(sa.reward - sb.reward);
|
|
1749
|
+
if (structuralDivergence || rewardGap >= minMargin) {
|
|
1750
|
+
divergenceIdx = k;
|
|
1751
|
+
break;
|
|
1752
|
+
}
|
|
1753
|
+
}
|
|
1754
|
+
if (divergenceIdx < 0) continue;
|
|
1755
|
+
if (divergenceIdx < minPrefix) continue;
|
|
1756
|
+
const aNext = a.steps[divergenceIdx];
|
|
1757
|
+
const bNext = b.steps[divergenceIdx];
|
|
1758
|
+
if (Math.abs(aNext.reward - bNext.reward) < minMargin) continue;
|
|
1759
|
+
const chosen = aNext.reward > bNext.reward ? aNext : bNext;
|
|
1760
|
+
const rejected = aNext.reward > bNext.reward ? bNext : aNext;
|
|
1761
|
+
const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId;
|
|
1762
|
+
const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId;
|
|
1763
|
+
triples.push({
|
|
1764
|
+
prefixRunId: chosenRun,
|
|
1765
|
+
prefixStepIndex: divergenceIdx - 1,
|
|
1766
|
+
chosenSpanId: chosen.spanId,
|
|
1767
|
+
chosenReward: chosen.reward,
|
|
1768
|
+
rejectedSpanId: rejected.spanId,
|
|
1769
|
+
rejectedReward: rejected.reward,
|
|
1770
|
+
rejectedRunId: rejectedRun,
|
|
1771
|
+
marginScore: chosen.reward - rejected.reward
|
|
1772
|
+
});
|
|
1773
|
+
}
|
|
1774
|
+
return triples;
|
|
1775
|
+
}
|
|
1776
|
+
//#endregion
|
|
1777
|
+
//#region src/rl/rl-campaign.ts
|
|
1778
|
+
/**
|
|
1779
|
+
* `runRLCampaign` — top-level orchestrator that runs the matrix and
|
|
1780
|
+
* produces every RL-ready artifact in one call.
|
|
1781
|
+
*
|
|
1782
|
+
* Wires:
|
|
1783
|
+
* 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)
|
|
1784
|
+
* 2. `extractVerifiableRewardsFromRecords` over the runs, separating deterministic
|
|
1785
|
+
* from probabilistic reward sources for the trainer
|
|
1786
|
+
* 3. `extractPreferences` to produce DPO/PPO/KTO triples
|
|
1787
|
+
* 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)
|
|
1788
|
+
* 5. `rubricPredictiveValidity` against an outcome store, when provided
|
|
1789
|
+
* 6. `detectRewardHacking` as a standing hygiene check
|
|
1790
|
+
* 7. Trainer-format export rows ready for prime-rl / TRL / verl
|
|
1791
|
+
*
|
|
1792
|
+
* The output `RLCampaignResult` is a single, audit-ready artifact: every
|
|
1793
|
+
* stage's output is in there. The consumer's downstream fits in a single
|
|
1794
|
+
* line: pass `result.preferences.pairs` to a DPO trainer,
|
|
1795
|
+
* `result.trainerRows.grpo` to GRPO, or `result.campaign.runs` plus
|
|
1796
|
+
* `result.rewardSignals` to a custom RL loop.
|
|
1797
|
+
*/
|
|
1220
1798
|
async function runRLCampaign(opts) {
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
|
-
|
|
1234
|
-
|
|
1235
|
-
|
|
1236
|
-
|
|
1237
|
-
|
|
1238
|
-
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
rewardSignals,
|
|
1278
|
-
preferences,
|
|
1279
|
-
interimConfidence,
|
|
1280
|
-
rewardHacking,
|
|
1281
|
-
predictiveValidity,
|
|
1282
|
-
trainerRows,
|
|
1283
|
-
summary,
|
|
1284
|
-
kind: "agent-eval-rl-campaign"
|
|
1285
|
-
};
|
|
1799
|
+
const splitTag = opts.splitTag ?? "search";
|
|
1800
|
+
const campaign = await runEvalCampaign({
|
|
1801
|
+
...opts,
|
|
1802
|
+
splitTag
|
|
1803
|
+
});
|
|
1804
|
+
const rewardSignals = extractVerifiableRewardsFromRecords(campaign.runs, opts.verifiableReward ?? {});
|
|
1805
|
+
const { rows: rolloutLines } = await mintRolloutRows(campaign.runs.filter((run) => runTaskScore(run) !== void 0), new InMemoryTraceStore());
|
|
1806
|
+
const preferences = extractPreferences(rolloutLines, {
|
|
1807
|
+
...opts.preferences,
|
|
1808
|
+
strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
|
|
1809
|
+
minMargin: opts.preferences?.minMargin ?? .05,
|
|
1810
|
+
split: opts.preferences?.split ?? splitTag
|
|
1811
|
+
});
|
|
1812
|
+
let interimConfidence = null;
|
|
1813
|
+
if (opts.report?.comparator) {
|
|
1814
|
+
const comparator = opts.report.comparator;
|
|
1815
|
+
const deltaSeries = collectPairedDeltaSeries(campaign.runs, comparator);
|
|
1816
|
+
if (deltaSeries.some((s) => s.deltas.length > 0)) interimConfidence = evaluateInterimReleaseConfidence({
|
|
1817
|
+
deltaSeries,
|
|
1818
|
+
alpha: opts.sequential?.alpha,
|
|
1819
|
+
bound: opts.sequential?.bound,
|
|
1820
|
+
rope: opts.sequential?.rope ?? opts.report?.rope
|
|
1821
|
+
});
|
|
1822
|
+
}
|
|
1823
|
+
const rewardHacking = detectRewardHacking({
|
|
1824
|
+
runs: campaign.runs,
|
|
1825
|
+
verifiableRewardOptions: opts.verifiableReward
|
|
1826
|
+
});
|
|
1827
|
+
let predictiveValidity = null;
|
|
1828
|
+
if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) predictiveValidity = await rubricPredictiveValidity({
|
|
1829
|
+
runs: campaign.runs,
|
|
1830
|
+
outcomes: opts.outcomeStore,
|
|
1831
|
+
outcomeMetrics: opts.outcomeMetrics
|
|
1832
|
+
});
|
|
1833
|
+
const trainerRows = {};
|
|
1834
|
+
if (opts.trainerExport?.dpo) trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, { lines: rolloutLines });
|
|
1835
|
+
if (opts.trainerExport?.grpo) trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo);
|
|
1836
|
+
if (opts.trainerExport?.sft) trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft);
|
|
1837
|
+
const summary = buildSummary({
|
|
1838
|
+
campaign,
|
|
1839
|
+
preferences,
|
|
1840
|
+
interimConfidence,
|
|
1841
|
+
rewardHacking,
|
|
1842
|
+
predictiveValidity
|
|
1843
|
+
});
|
|
1844
|
+
return {
|
|
1845
|
+
campaign,
|
|
1846
|
+
rewardSignals,
|
|
1847
|
+
preferences,
|
|
1848
|
+
interimConfidence,
|
|
1849
|
+
rewardHacking,
|
|
1850
|
+
predictiveValidity,
|
|
1851
|
+
trainerRows,
|
|
1852
|
+
summary,
|
|
1853
|
+
kind: "agent-eval-rl-campaign"
|
|
1854
|
+
};
|
|
1286
1855
|
}
|
|
1287
1856
|
function collectPairedDeltaSeries(runs, comparator) {
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1303
|
-
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
|
|
1857
|
+
const baseline = /* @__PURE__ */ new Map();
|
|
1858
|
+
for (const r of runs) {
|
|
1859
|
+
if (r.candidateId !== comparator) continue;
|
|
1860
|
+
const sid = r.scenarioId;
|
|
1861
|
+
const score = runTaskScore(r);
|
|
1862
|
+
if (score === void 0) continue;
|
|
1863
|
+
baseline.set(`${sid}::${r.seed}`, score);
|
|
1864
|
+
}
|
|
1865
|
+
const byCandidate = /* @__PURE__ */ new Map();
|
|
1866
|
+
for (const r of runs) {
|
|
1867
|
+
if (r.candidateId === comparator) continue;
|
|
1868
|
+
const sid = r.scenarioId;
|
|
1869
|
+
const score = runTaskScore(r);
|
|
1870
|
+
if (score === void 0) continue;
|
|
1871
|
+
const baseScore = baseline.get(`${sid}::${r.seed}`);
|
|
1872
|
+
if (typeof baseScore !== "number") continue;
|
|
1873
|
+
const arr = byCandidate.get(r.candidateId) ?? [];
|
|
1874
|
+
arr.push(score - baseScore);
|
|
1875
|
+
byCandidate.set(r.candidateId, arr);
|
|
1876
|
+
}
|
|
1877
|
+
return [...byCandidate.entries()].map(([candidateId, deltas]) => ({
|
|
1878
|
+
candidateId,
|
|
1879
|
+
deltas
|
|
1880
|
+
}));
|
|
1309
1881
|
}
|
|
1310
1882
|
function buildSummary(args) {
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
|
|
1331
|
-
|
|
1332
|
-
|
|
1333
|
-
|
|
1883
|
+
const c = args.campaign;
|
|
1884
|
+
const lines = [`${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}…)`, `preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`];
|
|
1885
|
+
if (args.interimConfidence) lines.push(`sequential verdict: ${args.interimConfidence.recommendation.decision}` + (args.interimConfidence.recommendation.candidateId ? ` ${args.interimConfidence.recommendation.candidateId}` : ""));
|
|
1886
|
+
lines.push(`reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`);
|
|
1887
|
+
if (args.predictiveValidity) {
|
|
1888
|
+
const top = args.predictiveValidity.ranked[0];
|
|
1889
|
+
lines.push(`top-rubric: ${top?.rubric ?? "none"} ρ=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? "no data"})`);
|
|
1890
|
+
}
|
|
1891
|
+
return lines.join(" | ");
|
|
1892
|
+
}
|
|
1893
|
+
//#endregion
|
|
1894
|
+
//#region src/rl/run-record-adapters.ts
|
|
1895
|
+
/**
|
|
1896
|
+
* Adapters: convert measurement outputs into the canonical `RunRecord[]`
|
|
1897
|
+
* artifact that `replayCache`, `pairedEvalueSequence`, and
|
|
1898
|
+
* `rubricPredictiveValidity` consume. Two sources:
|
|
1899
|
+
* - `campaignToRunRecords` — the campaign substrate's per-cell results
|
|
1900
|
+
* (the modern path: `runCampaign` / `runImprovementLoop` → records).
|
|
1901
|
+
* - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
|
|
1902
|
+
*
|
|
1903
|
+
* Adapters are thin and explicit — every mandatory `RunRecord` field comes
|
|
1904
|
+
* from a caller-supplied context (`commitSha`, `model`, `promptHash`,
|
|
1905
|
+
* `configHash`) plus the cell's runtime data. The validator still rejects
|
|
1906
|
+
* bare-alias model strings — the caller snapshot-pins.
|
|
1907
|
+
*/
|
|
1908
|
+
/**
|
|
1909
|
+
* Convert a `CampaignResult` into canonical `RunRecord[]`, one per cell.
|
|
1910
|
+
* Successful judged cells carry their mean judge composite and dimensions.
|
|
1911
|
+
* Errored or unjudged cells remain unlabeled while retaining explicit terminal
|
|
1912
|
+
* outcome, execution-error count, token usage, cost, and failure detail.
|
|
1913
|
+
* `candidateId` identifies the measured surface and defaults to the campaign
|
|
1914
|
+
* manifest hash.
|
|
1915
|
+
*/
|
|
1334
1916
|
function campaignToRunRecords(campaign, ctx) {
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
|
|
1349
|
-
|
|
1350
|
-
|
|
1917
|
+
const splitTag = ctx.splitTag ?? "search";
|
|
1918
|
+
const candidateId = ctx.candidateId ?? campaign.manifestHash;
|
|
1919
|
+
return campaign.cells.map((cell) => campaignCellToRunRecord(cell, {
|
|
1920
|
+
runId: cell.cellId,
|
|
1921
|
+
experimentId: ctx.experimentId,
|
|
1922
|
+
candidateId,
|
|
1923
|
+
model: ctx.model,
|
|
1924
|
+
promptHash: ctx.promptHash,
|
|
1925
|
+
configHash: ctx.configHash,
|
|
1926
|
+
commitSha: ctx.commitSha,
|
|
1927
|
+
splitTag,
|
|
1928
|
+
defaultCostUsd: ctx.defaultCostUsd
|
|
1929
|
+
}));
|
|
1930
|
+
}
|
|
1931
|
+
/**
|
|
1932
|
+
* Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
|
|
1933
|
+
* A split score is emitted only when `report.taskScore` proves the configured
|
|
1934
|
+
* scoring panel completed. Partial scores remain in `outcome.raw` for
|
|
1935
|
+
* diagnosis. Layer errors and timeouts become judge or execution telemetry;
|
|
1936
|
+
* only a scored `fail` layer may produce task-failure detail.
|
|
1937
|
+
*/
|
|
1351
1938
|
function verificationReportToRunRecord(report, ctx, opts = {}) {
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1361
|
-
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1385
|
-
|
|
1386
|
-
|
|
1387
|
-
|
|
1388
|
-
|
|
1389
|
-
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1939
|
+
const splitTag = ctx.splitTag ?? "search";
|
|
1940
|
+
const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
|
|
1941
|
+
const taskScore = report.layers.some(hasValidTaskMeasurement) && isValidScore(report.taskScore) ? report.taskScore : void 0;
|
|
1942
|
+
let executionErrorCount = 0;
|
|
1943
|
+
let judgeErrorCount = 0;
|
|
1944
|
+
let layerErrorCount = 0;
|
|
1945
|
+
let layerTimeoutCount = 0;
|
|
1946
|
+
let unscoredLayerCount = 0;
|
|
1947
|
+
const raw = {
|
|
1948
|
+
pass_count: report.passCount,
|
|
1949
|
+
fail_count: report.failCount,
|
|
1950
|
+
error_count: report.errorCount,
|
|
1951
|
+
skipped_count: report.skippedCount,
|
|
1952
|
+
duration_ms: report.durationMs,
|
|
1953
|
+
execution_error_count: 0
|
|
1954
|
+
};
|
|
1955
|
+
for (const layer of report.layers) {
|
|
1956
|
+
if (hasValidTaskMeasurement(layer)) raw[`layer.${layer.layer}`] = layer.score;
|
|
1957
|
+
else unscoredLayerCount++;
|
|
1958
|
+
raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
|
|
1959
|
+
if (layer.status === "error" || layer.status === "timeout") {
|
|
1960
|
+
if (layer.errorSource === "judge") judgeErrorCount++;
|
|
1961
|
+
else executionErrorCount++;
|
|
1962
|
+
if (layer.status === "error") layerErrorCount++;
|
|
1963
|
+
else layerTimeoutCount++;
|
|
1964
|
+
}
|
|
1965
|
+
if (layer.diagnostics) {
|
|
1966
|
+
for (const [k, v] of Object.entries(layer.diagnostics)) if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
|
|
1967
|
+
}
|
|
1968
|
+
}
|
|
1969
|
+
raw.execution_error_count = executionErrorCount;
|
|
1970
|
+
if (judgeErrorCount > 0) raw.judge_error_count = judgeErrorCount;
|
|
1971
|
+
if (layerErrorCount > 0) raw.layer_error_count = layerErrorCount;
|
|
1972
|
+
if (layerTimeoutCount > 0) raw.layer_timeout_count = layerTimeoutCount;
|
|
1973
|
+
if (unscoredLayerCount > 0) raw.unscored_layer_count = unscoredLayerCount;
|
|
1974
|
+
if (taskScore !== void 0) raw.blended_score = taskScore;
|
|
1975
|
+
const firstScoredFailure = report.layers.find((layer) => layer.status === "fail" && hasValidTaskMeasurement(layer));
|
|
1976
|
+
const outcome = { raw };
|
|
1977
|
+
if (taskScore !== void 0) if (splitTag === "holdout") outcome.holdoutScore = taskScore;
|
|
1978
|
+
else outcome.searchScore = taskScore;
|
|
1979
|
+
return {
|
|
1980
|
+
runId,
|
|
1981
|
+
experimentId: ctx.experimentId,
|
|
1982
|
+
candidateId: ctx.candidateId,
|
|
1983
|
+
seed: 0,
|
|
1984
|
+
model: ctx.model,
|
|
1985
|
+
promptHash: ctx.promptHash,
|
|
1986
|
+
configHash: ctx.configHash,
|
|
1987
|
+
commitSha: ctx.commitSha,
|
|
1988
|
+
wallMs: report.durationMs,
|
|
1989
|
+
costUsd: ctx.defaultCostUsd ?? null,
|
|
1990
|
+
costProvenance: ctx.defaultCostUsd === void 0 ? {
|
|
1991
|
+
kind: "uncaptured",
|
|
1992
|
+
usd: null
|
|
1993
|
+
} : {
|
|
1994
|
+
kind: "estimated",
|
|
1995
|
+
usd: ctx.defaultCostUsd
|
|
1996
|
+
},
|
|
1997
|
+
tokenUsage: {
|
|
1998
|
+
input: 0,
|
|
1999
|
+
output: 0
|
|
2000
|
+
},
|
|
2001
|
+
terminalOutcome: "succeeded",
|
|
2002
|
+
outcome,
|
|
2003
|
+
...firstScoredFailure ? {
|
|
2004
|
+
failureClass: "unknown",
|
|
2005
|
+
failureMode: `layer_${firstScoredFailure.layer}_fail`
|
|
2006
|
+
} : {},
|
|
2007
|
+
splitTag,
|
|
2008
|
+
scenarioId: ctx.scenarioId
|
|
2009
|
+
};
|
|
1421
2010
|
}
|
|
1422
2011
|
function hasValidTaskMeasurement(layer) {
|
|
1423
|
-
|
|
2012
|
+
return (layer.status === "pass" || layer.status === "fail") && isValidScore(layer.score);
|
|
1424
2013
|
}
|
|
1425
2014
|
function isValidScore(score) {
|
|
1426
|
-
|
|
1427
|
-
}
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
2015
|
+
return typeof score === "number" && Number.isFinite(score) && score >= 0 && score <= 1;
|
|
2016
|
+
}
|
|
2017
|
+
//#endregion
|
|
2018
|
+
//#region src/rl/sim-fidelity.ts
|
|
2019
|
+
/**
|
|
2020
|
+
* Simulator fidelity — score a user SIMULATOR's realism against real-user
|
|
2021
|
+
* trace distributions.
|
|
2022
|
+
*
|
|
2023
|
+
* Synthetic-persona evals (`PersonaConfig`-driven canonical evals, fuzz
|
|
2024
|
+
* user-simulator objectives) stand in for real users in most of the numbers
|
|
2025
|
+
* we publish. The standing threat is the Sim2Real gap: a simulator that is
|
|
2026
|
+
* distributionally unlike production creates "easy mode" and silently
|
|
2027
|
+
* inflates every score built on it. This module measures that gap from the
|
|
2028
|
+
* SAME artifact both sides already produce — `RunRecord`s — so no new
|
|
2029
|
+
* capture pipeline is needed:
|
|
2030
|
+
*
|
|
2031
|
+
* - `simFidelityReport` — per-feature Jensen-Shannon divergence between
|
|
2032
|
+
* simulated and production record distributions, collapsed into a
|
|
2033
|
+
* fidelity coefficient in [0,1].
|
|
2034
|
+
* - `easyModeCheck` — the headline academic failure mode (sim inflates
|
|
2035
|
+
* pass-rate over production) as its own named artifact.
|
|
2036
|
+
*
|
|
2037
|
+
* Every synthetic-persona eval result should publish its fidelity
|
|
2038
|
+
* coefficient alongside the score — a number from an unrepresentative
|
|
2039
|
+
* simulator is an unlabeled estimate. Wire-in points:
|
|
2040
|
+
*
|
|
2041
|
+
* - canonical persona evals: pass the campaign's `RunRecord`s as
|
|
2042
|
+
* `simulated` and intake-adapter output (`contract/intake`: OTel spans,
|
|
2043
|
+
* feedback tables, coding-agent sessions) as `production`
|
|
2044
|
+
* - the fuzz user-sim objective: use `1 - report.fidelity` as a realism
|
|
2045
|
+
* penalty when searching over generated personas
|
|
2046
|
+
* - the durable corpus (`./corpus`): both sides read straight from
|
|
2047
|
+
* `readCorpus` — tag sim vs production by `experimentId`
|
|
2048
|
+
*/
|
|
2049
|
+
/** Reserved histogram category for `null` feature values. A capture-rate
|
|
2050
|
+
* difference (one side instruments a signal, the other does not) registers
|
|
2051
|
+
* as divergence by design: a simulator that produces no tool traces is not
|
|
2052
|
+
* representative of production that does. */
|
|
2053
|
+
const ABSENT_CATEGORY = "(absent)";
|
|
2054
|
+
/** Minimum non-null observations PER SIDE for a feature to enter the
|
|
2055
|
+
* fidelity mean. Below this the JSD estimate is sampling noise. */
|
|
2056
|
+
const DEFAULT_MIN_N_PER_FEATURE = 20;
|
|
2057
|
+
/** Quantile buckets used to discretize numeric features. Quartiles balance
|
|
2058
|
+
* resolution against per-bucket sample size at the default minN. */
|
|
2059
|
+
const DEFAULT_QUANTILE_BUCKETS = 4;
|
|
2060
|
+
/** Fidelity at or above this → 'representative'; below → 'skewed'.
|
|
2061
|
+
* 1 − 0.8 = mean JSD 0.2 ≈ distributions that mostly overlap with one
|
|
2062
|
+
* clearly shifted mode — the point where per-feature shifts start changing
|
|
2063
|
+
* which failure classes an eval can even observe. */
|
|
2064
|
+
const REPRESENTATIVE_MIN_FIDELITY = .8;
|
|
2065
|
+
const TOP_SHIFT_COUNT = 5;
|
|
2066
|
+
/**
|
|
2067
|
+
* Default feature set — ONLY fields verified present on both simulated and
|
|
2068
|
+
* production records:
|
|
2069
|
+
*
|
|
2070
|
+
* - `score`, `wall_ms`, `output_tokens` — mandatory per the `RunRecord`
|
|
2071
|
+
* validator (non-finite values read as absent rather than poisoning a
|
|
2072
|
+
* bucket).
|
|
2073
|
+
* - `failure_class` — optional taxonomy field; absent counted explicitly.
|
|
2074
|
+
* - `turn_count`, `tool_errors`, `tool_error_recovery` — derived from the
|
|
2075
|
+
* `outcome.raw` counters the intake adapters and eval harnesses write
|
|
2076
|
+
* (`turns_completed`, `assistant_messages`, `tool_errors`,
|
|
2077
|
+
* `turns_aborted`); absent on records whose producer did not capture
|
|
2078
|
+
* them, counted explicitly.
|
|
2079
|
+
* - `completion_length` — from the optional `CorpusRecord` trajectory
|
|
2080
|
+
* text; the message-length proxy when records come from the corpus.
|
|
2081
|
+
*
|
|
2082
|
+
* `RunRecord` carries event COUNTS, not event ordering, so
|
|
2083
|
+
* `tool_error_recovery` is a counts-only derivation: errors occurred and the
|
|
2084
|
+
* run still completed cleanly ('recovered') vs aborted or classified as a
|
|
2085
|
+
* failure ('unrecovered') — not a literal error→retry sequence check.
|
|
2086
|
+
*/
|
|
2087
|
+
const defaultBehaviorFeatures = (record) => {
|
|
2088
|
+
const raw = record.outcome?.raw ?? {};
|
|
2089
|
+
const toolErrors = finiteOrNull(raw.tool_errors);
|
|
2090
|
+
const turnsAborted = finiteOrNull(raw.turns_aborted);
|
|
2091
|
+
const completion = record.completion;
|
|
2092
|
+
return {
|
|
2093
|
+
score: finiteOrNull(observedSplitScore(record, "holdout")) ?? finiteOrNull(observedSplitScore(record, "search")),
|
|
2094
|
+
failure_class: record.failureClass ?? null,
|
|
2095
|
+
wall_ms: finiteOrNull(record.wallMs),
|
|
2096
|
+
output_tokens: finiteOrNull(record.tokenUsage?.output),
|
|
2097
|
+
turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),
|
|
2098
|
+
tool_errors: toolErrors,
|
|
2099
|
+
tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),
|
|
2100
|
+
completion_length: typeof completion === "string" ? completion.length : null
|
|
2101
|
+
};
|
|
1450
2102
|
};
|
|
1451
2103
|
function toolErrorRecovery(toolErrors, turnsAborted, failureClass) {
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
return failed ? "unrecovered" : "recovered";
|
|
2104
|
+
if (toolErrors === null) return null;
|
|
2105
|
+
if (toolErrors === 0) return "no-tool-errors";
|
|
2106
|
+
return (turnsAborted ?? 0) > 0 || failureClass !== void 0 && failureClass !== "success" ? "unrecovered" : "recovered";
|
|
1456
2107
|
}
|
|
1457
2108
|
function finiteOrNull(value) {
|
|
1458
|
-
|
|
1459
|
-
}
|
|
2109
|
+
return typeof value === "number" && Number.isFinite(value) ? value : null;
|
|
2110
|
+
}
|
|
2111
|
+
/**
|
|
2112
|
+
* Jensen-Shannon divergence between two categorical histograms (raw counts;
|
|
2113
|
+
* normalized internally). Log base 2 → bounded [0,1]: 0 = identical
|
|
2114
|
+
* distributions, 1 = disjoint support. Symmetric, defined even where the
|
|
2115
|
+
* supports differ — exactly the regime sim-vs-production comparison lives in.
|
|
2116
|
+
* Throws on zero-mass or negative/non-finite counts: an empty histogram has
|
|
2117
|
+
* no distribution and a silent 0 would read as "perfectly representative".
|
|
2118
|
+
*/
|
|
1460
2119
|
function jsDivergence(p, q) {
|
|
1461
|
-
|
|
1462
|
-
|
|
1463
|
-
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
function quantileEdges(values, bucketCount =
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
edges.push(lo + (pos - Math.floor(pos)) * (hi - lo));
|
|
1505
|
-
}
|
|
1506
|
-
return [...new Set(edges)];
|
|
1507
|
-
}
|
|
2120
|
+
const keys = /* @__PURE__ */ new Set([...Object.keys(p), ...Object.keys(q)]);
|
|
2121
|
+
if (keys.size === 0) throw new ValidationError("jsDivergence: both histograms are empty");
|
|
2122
|
+
let pSum = 0;
|
|
2123
|
+
let qSum = 0;
|
|
2124
|
+
for (const key of keys) {
|
|
2125
|
+
const pv = p[key] ?? 0;
|
|
2126
|
+
const qv = q[key] ?? 0;
|
|
2127
|
+
if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) throw new ValidationError(`jsDivergence: negative or non-finite count for category "${key}"`);
|
|
2128
|
+
pSum += pv;
|
|
2129
|
+
qSum += qv;
|
|
2130
|
+
}
|
|
2131
|
+
if (pSum === 0 || qSum === 0) throw new ValidationError("jsDivergence: a histogram with zero total mass has no distribution");
|
|
2132
|
+
let divergence = 0;
|
|
2133
|
+
for (const key of keys) {
|
|
2134
|
+
const pp = (p[key] ?? 0) / pSum;
|
|
2135
|
+
const qp = (q[key] ?? 0) / qSum;
|
|
2136
|
+
const m = (pp + qp) / 2;
|
|
2137
|
+
if (pp > 0) divergence += .5 * pp * Math.log2(pp / m);
|
|
2138
|
+
if (qp > 0) divergence += .5 * qp * Math.log2(qp / m);
|
|
2139
|
+
}
|
|
2140
|
+
return Math.min(1, Math.max(0, divergence));
|
|
2141
|
+
}
|
|
2142
|
+
/**
|
|
2143
|
+
* Deterministic quantile edges over a value set (the UNION of both sides, so
|
|
2144
|
+
* sim and production land in the same buckets). Linear interpolation between
|
|
2145
|
+
* order statistics; duplicate edges from heavy ties collapse into fewer,
|
|
2146
|
+
* wider buckets. Returns `bucketCount - 1` edges before deduplication.
|
|
2147
|
+
*/
|
|
2148
|
+
function quantileEdges(values, bucketCount = 4) {
|
|
2149
|
+
if (values.length === 0) throw new ValidationError("quantileEdges: requires at least one value");
|
|
2150
|
+
if (!Number.isInteger(bucketCount) || bucketCount < 2) throw new ValidationError(`quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`);
|
|
2151
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
2152
|
+
const edges = [];
|
|
2153
|
+
for (let k = 1; k < bucketCount; k++) {
|
|
2154
|
+
const pos = k / bucketCount * (sorted.length - 1);
|
|
2155
|
+
const lo = sorted[Math.floor(pos)];
|
|
2156
|
+
const hi = sorted[Math.ceil(pos)];
|
|
2157
|
+
edges.push(lo + (pos - Math.floor(pos)) * (hi - lo));
|
|
2158
|
+
}
|
|
2159
|
+
return [...new Set(edges)];
|
|
2160
|
+
}
|
|
2161
|
+
/** Stable half-open bucket label for a value against quantile edges:
|
|
2162
|
+
* `[-inf,e0)`, `[e0,e1)`, …, `[eLast,+inf)`. */
|
|
1508
2163
|
function bucketLabel(value, edges) {
|
|
1509
|
-
|
|
1510
|
-
|
|
1511
|
-
|
|
1512
|
-
|
|
1513
|
-
|
|
1514
|
-
|
|
2164
|
+
let i = 0;
|
|
2165
|
+
while (i < edges.length && value >= edges[i]) i++;
|
|
2166
|
+
return `[${i === 0 ? "-inf" : String(edges[i - 1])},${i === edges.length ? "+inf" : String(edges[i])})`;
|
|
2167
|
+
}
|
|
2168
|
+
/**
|
|
2169
|
+
* Compare a simulator's RunRecords against production RunRecords, feature by
|
|
2170
|
+
* feature. Numeric features are bucketed by deterministic quantiles of the
|
|
2171
|
+
* union; nulls count as an explicit `ABSENT_CATEGORY`. Throws on empty
|
|
2172
|
+
* inputs — "no records" is a wiring error, not a distribution.
|
|
2173
|
+
*/
|
|
1515
2174
|
function simFidelityReport(simulated, production, opts = {}) {
|
|
1516
|
-
|
|
1517
|
-
|
|
1518
|
-
|
|
1519
|
-
|
|
1520
|
-
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
|
|
1524
|
-
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1542
|
-
|
|
1543
|
-
|
|
1544
|
-
|
|
1545
|
-
|
|
1546
|
-
|
|
1547
|
-
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1560
|
-
|
|
1561
|
-
perDimension,
|
|
1562
|
-
fidelity,
|
|
1563
|
-
insufficientData,
|
|
1564
|
-
verdict: fidelity >= REPRESENTATIVE_MIN_FIDELITY ? "representative" : "skewed"
|
|
1565
|
-
};
|
|
2175
|
+
if (simulated.length === 0) throw new ValidationError("simFidelityReport: simulated records are empty");
|
|
2176
|
+
if (production.length === 0) throw new ValidationError("simFidelityReport: production records are empty");
|
|
2177
|
+
const extract = opts.features ?? defaultBehaviorFeatures;
|
|
2178
|
+
const minN = opts.minNPerFeature ?? 20;
|
|
2179
|
+
const simMaps = simulated.map(extract);
|
|
2180
|
+
const prodMaps = production.map(extract);
|
|
2181
|
+
const featureNames = [];
|
|
2182
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2183
|
+
for (const map of [...simMaps, ...prodMaps]) for (const name of Object.keys(map)) if (!seen.has(name)) {
|
|
2184
|
+
seen.add(name);
|
|
2185
|
+
featureNames.push(name);
|
|
2186
|
+
}
|
|
2187
|
+
const perDimension = [];
|
|
2188
|
+
const insufficientData = [];
|
|
2189
|
+
for (const feature of featureNames) {
|
|
2190
|
+
const simVals = simMaps.map((m) => m[feature] ?? null);
|
|
2191
|
+
const prodVals = prodMaps.map((m) => m[feature] ?? null);
|
|
2192
|
+
const nSim = simVals.filter((v) => v !== null).length;
|
|
2193
|
+
const nProd = prodVals.filter((v) => v !== null).length;
|
|
2194
|
+
if (nSim < minN || nProd < minN) {
|
|
2195
|
+
insufficientData.push(feature);
|
|
2196
|
+
continue;
|
|
2197
|
+
}
|
|
2198
|
+
const { sim, prod } = histograms(feature, simVals, prodVals);
|
|
2199
|
+
perDimension.push({
|
|
2200
|
+
feature,
|
|
2201
|
+
divergence: jsDivergence(sim, prod),
|
|
2202
|
+
topShifts: topShifts(sim, simVals.length, prod, prodVals.length),
|
|
2203
|
+
nSim,
|
|
2204
|
+
nProd
|
|
2205
|
+
});
|
|
2206
|
+
}
|
|
2207
|
+
if (perDimension.length === 0) return {
|
|
2208
|
+
perDimension,
|
|
2209
|
+
fidelity: NaN,
|
|
2210
|
+
insufficientData,
|
|
2211
|
+
verdict: "insufficient-data"
|
|
2212
|
+
};
|
|
2213
|
+
const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length;
|
|
2214
|
+
return {
|
|
2215
|
+
perDimension,
|
|
2216
|
+
fidelity,
|
|
2217
|
+
insufficientData,
|
|
2218
|
+
verdict: fidelity >= .8 ? "representative" : "skewed"
|
|
2219
|
+
};
|
|
1566
2220
|
}
|
|
1567
2221
|
function histograms(feature, simVals, prodVals) {
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
|
|
1579
|
-
|
|
1580
|
-
|
|
1581
|
-
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1587
|
-
|
|
1588
|
-
|
|
1589
|
-
|
|
1590
|
-
for (const v of vals) {
|
|
1591
|
-
const key = v === null ? ABSENT_CATEGORY : toCategory(v);
|
|
1592
|
-
hist[key] = (hist[key] ?? 0) + 1;
|
|
1593
|
-
}
|
|
1594
|
-
return hist;
|
|
1595
|
-
};
|
|
1596
|
-
return { sim: count(simVals), prod: count(prodVals) };
|
|
2222
|
+
const kinds = /* @__PURE__ */ new Set();
|
|
2223
|
+
for (const v of [...simVals, ...prodVals]) if (v !== null) kinds.add(typeof v);
|
|
2224
|
+
if (kinds.size > 1) throw new ValidationError(`simFidelityReport: feature "${feature}" mixes string and number values — an extractor must return one kind per feature`);
|
|
2225
|
+
let toCategory;
|
|
2226
|
+
if (kinds.has("number")) {
|
|
2227
|
+
const union = [];
|
|
2228
|
+
for (const v of [...simVals, ...prodVals]) if (v !== null) union.push(v);
|
|
2229
|
+
const edges = quantileEdges(union);
|
|
2230
|
+
toCategory = (v) => bucketLabel(v, edges);
|
|
2231
|
+
} else toCategory = (v) => v;
|
|
2232
|
+
const count = (vals) => {
|
|
2233
|
+
const hist = {};
|
|
2234
|
+
for (const v of vals) {
|
|
2235
|
+
const key = v === null ? ABSENT_CATEGORY : toCategory(v);
|
|
2236
|
+
hist[key] = (hist[key] ?? 0) + 1;
|
|
2237
|
+
}
|
|
2238
|
+
return hist;
|
|
2239
|
+
};
|
|
2240
|
+
return {
|
|
2241
|
+
sim: count(simVals),
|
|
2242
|
+
prod: count(prodVals)
|
|
2243
|
+
};
|
|
1597
2244
|
}
|
|
1598
2245
|
function topShifts(sim, simTotal, prod, prodTotal) {
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
2246
|
+
const shifts = [.../* @__PURE__ */ new Set([...Object.keys(sim), ...Object.keys(prod)])].map((value) => ({
|
|
2247
|
+
value,
|
|
2248
|
+
pSim: (sim[value] ?? 0) / simTotal,
|
|
2249
|
+
pProd: (prod[value] ?? 0) / prodTotal
|
|
2250
|
+
}));
|
|
2251
|
+
shifts.sort((a, b) => {
|
|
2252
|
+
const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd);
|
|
2253
|
+
return delta !== 0 ? delta : a.value.localeCompare(b.value);
|
|
2254
|
+
});
|
|
2255
|
+
return shifts.slice(0, TOP_SHIFT_COUNT);
|
|
2256
|
+
}
|
|
2257
|
+
/**
|
|
2258
|
+
* The headline simulator failure mode as its own named artifact: a simulator
|
|
2259
|
+
* that creates "easy mode" inflates pass-rate relative to production, and
|
|
2260
|
+
* every score measured against it overstates reality. Throws on empty inputs
|
|
2261
|
+
* and on records carrying neither score — a silently-skipped record would
|
|
2262
|
+
* bias the very rate this check exists to keep honest.
|
|
2263
|
+
*/
|
|
1611
2264
|
function easyModeCheck(simulated, production, opts = {}) {
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
|
|
1622
|
-
|
|
1623
|
-
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
2265
|
+
if (simulated.length === 0) throw new ValidationError("easyModeCheck: simulated records are empty");
|
|
2266
|
+
if (production.length === 0) throw new ValidationError("easyModeCheck: production records are empty");
|
|
2267
|
+
const threshold = opts.passThreshold ?? .5;
|
|
2268
|
+
const tolerance = opts.inflationTolerance ?? .1;
|
|
2269
|
+
const passRate = (records, side) => {
|
|
2270
|
+
let passes = 0;
|
|
2271
|
+
for (const r of records) {
|
|
2272
|
+
const score = finiteOrNull(observedSplitScore(r, "holdout")) ?? finiteOrNull(observedSplitScore(r, "search"));
|
|
2273
|
+
if (score === null) throw new ValidationError(`easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`);
|
|
2274
|
+
if (score >= threshold) passes++;
|
|
2275
|
+
}
|
|
2276
|
+
return passes / records.length;
|
|
2277
|
+
};
|
|
2278
|
+
const simPassRate = passRate(simulated, "simulated");
|
|
2279
|
+
const prodPassRate = passRate(production, "production");
|
|
2280
|
+
const gap = simPassRate - prodPassRate;
|
|
2281
|
+
return {
|
|
2282
|
+
simPassRate,
|
|
2283
|
+
prodPassRate,
|
|
2284
|
+
gap,
|
|
2285
|
+
inflated: gap > tolerance
|
|
2286
|
+
};
|
|
2287
|
+
}
|
|
2288
|
+
//#endregion
|
|
2289
|
+
//#region src/rl/tournament.ts
|
|
2290
|
+
/**
|
|
2291
|
+
* Bradley-Terry MLE via Hunter's MM algorithm.
|
|
2292
|
+
*
|
|
2293
|
+
* Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
|
|
2294
|
+
* where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
|
|
2295
|
+
*
|
|
2296
|
+
* Returns log-strengths normalized so the smallest is 0 (any constant
|
|
2297
|
+
* offset is unobservable in BT — only differences are identified).
|
|
2298
|
+
*/
|
|
1640
2299
|
function fitBradleyTerry(outcomes, opts = {}) {
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
|
|
1664
|
-
|
|
1665
|
-
|
|
1666
|
-
|
|
1667
|
-
|
|
1668
|
-
|
|
1669
|
-
|
|
1670
|
-
|
|
1671
|
-
|
|
1672
|
-
|
|
1673
|
-
|
|
1674
|
-
|
|
1675
|
-
|
|
1676
|
-
|
|
1677
|
-
|
|
1678
|
-
|
|
1679
|
-
|
|
1680
|
-
|
|
1681
|
-
|
|
1682
|
-
|
|
1683
|
-
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1690
|
-
|
|
1691
|
-
|
|
1692
|
-
|
|
1693
|
-
|
|
1694
|
-
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1702
|
-
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
1715
|
-
|
|
1716
|
-
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1721
|
-
|
|
1722
|
-
|
|
1723
|
-
|
|
1724
|
-
|
|
2300
|
+
const tol = opts.tolerance ?? 1e-6;
|
|
2301
|
+
const maxIter = opts.maxIterations ?? 256;
|
|
2302
|
+
const smoothing = opts.smoothing ?? .1;
|
|
2303
|
+
const candidates = /* @__PURE__ */ new Set();
|
|
2304
|
+
for (const o of outcomes) {
|
|
2305
|
+
candidates.add(o.winner);
|
|
2306
|
+
candidates.add(o.loser);
|
|
2307
|
+
}
|
|
2308
|
+
const ids = [...candidates].sort();
|
|
2309
|
+
const idx = new Map(ids.map((id, i) => [id, i]));
|
|
2310
|
+
const n = ids.length;
|
|
2311
|
+
if (n === 0) return {
|
|
2312
|
+
ratings: [],
|
|
2313
|
+
iterations: 0,
|
|
2314
|
+
finalDelta: 0,
|
|
2315
|
+
converged: true
|
|
2316
|
+
};
|
|
2317
|
+
if (n === 1) return {
|
|
2318
|
+
ratings: [{
|
|
2319
|
+
candidateId: ids[0],
|
|
2320
|
+
strength: 1,
|
|
2321
|
+
logStrength: 0,
|
|
2322
|
+
n: 0,
|
|
2323
|
+
wins: 0
|
|
2324
|
+
}],
|
|
2325
|
+
iterations: 0,
|
|
2326
|
+
finalDelta: 0,
|
|
2327
|
+
converged: true
|
|
2328
|
+
};
|
|
2329
|
+
const W = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
2330
|
+
const N = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
2331
|
+
for (const o of outcomes) {
|
|
2332
|
+
const i = idx.get(o.winner);
|
|
2333
|
+
const j = idx.get(o.loser);
|
|
2334
|
+
const w = o.weight ?? 1;
|
|
2335
|
+
if (o.draw) {
|
|
2336
|
+
W[i][j] += .5 * w;
|
|
2337
|
+
W[j][i] += .5 * w;
|
|
2338
|
+
} else W[i][j] += w;
|
|
2339
|
+
N[i][j] += w;
|
|
2340
|
+
N[j][i] += w;
|
|
2341
|
+
}
|
|
2342
|
+
const winsTotal = new Array(n).fill(0);
|
|
2343
|
+
for (let i = 0; i < n; i++) {
|
|
2344
|
+
for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
|
|
2345
|
+
winsTotal[i] += smoothing;
|
|
2346
|
+
}
|
|
2347
|
+
const compsTotal = new Array(n).fill(0);
|
|
2348
|
+
for (let i = 0; i < n; i++) for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
|
|
2349
|
+
let theta = new Array(n).fill(1);
|
|
2350
|
+
let iter = 0;
|
|
2351
|
+
let delta = Infinity;
|
|
2352
|
+
for (; iter < maxIter; iter++) {
|
|
2353
|
+
const newTheta = new Array(n);
|
|
2354
|
+
for (let i = 0; i < n; i++) {
|
|
2355
|
+
let denom = 0;
|
|
2356
|
+
for (let j = 0; j < n; j++) {
|
|
2357
|
+
if (j === i) continue;
|
|
2358
|
+
if (N[i][j] === 0) continue;
|
|
2359
|
+
denom += N[i][j] / (theta[i] + theta[j]);
|
|
2360
|
+
}
|
|
2361
|
+
newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
|
|
2362
|
+
}
|
|
2363
|
+
let logSum = 0;
|
|
2364
|
+
for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
|
|
2365
|
+
const norm = Math.exp(logSum / n);
|
|
2366
|
+
for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
|
|
2367
|
+
delta = 0;
|
|
2368
|
+
for (let i = 0; i < n; i++) {
|
|
2369
|
+
const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
|
|
2370
|
+
if (d > delta) delta = d;
|
|
2371
|
+
}
|
|
2372
|
+
theta = newTheta;
|
|
2373
|
+
if (delta < tol) break;
|
|
2374
|
+
}
|
|
2375
|
+
const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
|
|
2376
|
+
return {
|
|
2377
|
+
ratings: ids.map((id, i) => ({
|
|
2378
|
+
candidateId: id,
|
|
2379
|
+
strength: theta[i],
|
|
2380
|
+
logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
|
|
2381
|
+
n: compsTotal[i],
|
|
2382
|
+
wins: winsTotal[i] - smoothing
|
|
2383
|
+
})).sort((a, b) => b.strength - a.strength),
|
|
2384
|
+
iterations: iter,
|
|
2385
|
+
finalDelta: delta,
|
|
2386
|
+
converged: delta < tol
|
|
2387
|
+
};
|
|
1725
2388
|
}
|
|
1726
2389
|
function applyEloUpdate(ratings, outcome, opts = {}) {
|
|
1727
|
-
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
1737
|
-
|
|
1738
|
-
|
|
1739
|
-
|
|
2390
|
+
const defaultRating = opts.defaultRating ?? 1500;
|
|
2391
|
+
const k = opts.kFactor ?? 32;
|
|
2392
|
+
const rW = ratings.get(outcome.winner) ?? defaultRating;
|
|
2393
|
+
const rL = ratings.get(outcome.loser) ?? defaultRating;
|
|
2394
|
+
const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
|
|
2395
|
+
const scoreW = outcome.draw ? .5 : 1;
|
|
2396
|
+
const scoreL = outcome.draw ? .5 : 0;
|
|
2397
|
+
const w = outcome.weight ?? 1;
|
|
2398
|
+
const winnerDelta = k * w * (scoreW - expectedW);
|
|
2399
|
+
const loserDelta = k * w * (scoreL - (1 - expectedW));
|
|
2400
|
+
ratings.set(outcome.winner, rW + winnerDelta);
|
|
2401
|
+
ratings.set(outcome.loser, rL + loserDelta);
|
|
2402
|
+
return {
|
|
2403
|
+
winnerDelta,
|
|
2404
|
+
loserDelta
|
|
2405
|
+
};
|
|
1740
2406
|
}
|
|
1741
2407
|
function buildPairwiseFromCampaign(input) {
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1769
|
-
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
|
|
1778
|
-
bestOfN,
|
|
1779
|
-
bucketLabel,
|
|
1780
|
-
buildDatasetFromCorpus,
|
|
1781
|
-
buildPairwiseFromCampaign,
|
|
1782
|
-
buildRlDataset,
|
|
1783
|
-
campaignToRunRecords,
|
|
1784
|
-
compareAdaptationCurves,
|
|
1785
|
-
datasheetToMarkdown,
|
|
1786
|
-
defaultBehaviorFeatures,
|
|
1787
|
-
detectRewardHacking,
|
|
1788
|
-
doublyRobust,
|
|
1789
|
-
easyModeCheck,
|
|
1790
|
-
extractPreferences,
|
|
1791
|
-
extractStepRewards,
|
|
1792
|
-
extractVerifiableReward,
|
|
1793
|
-
extractVerifiableRewardsFromRecords,
|
|
1794
|
-
filterDeterministicallyRewarded,
|
|
1795
|
-
firstPassK,
|
|
1796
|
-
fitBradleyTerry,
|
|
1797
|
-
injectIrrelevantClause,
|
|
1798
|
-
inverseProbabilityWeighting,
|
|
1799
|
-
isTrainingRunEligible,
|
|
1800
|
-
jsDivergence,
|
|
1801
|
-
observationsFromRunRecords,
|
|
1802
|
-
offPolicyEstimateAll,
|
|
1803
|
-
paretoFrontier,
|
|
1804
|
-
prmTrainingPairs,
|
|
1805
|
-
quantileEdges,
|
|
1806
|
-
readCorpus,
|
|
1807
|
-
renameVariables,
|
|
1808
|
-
runAdaptationCurve,
|
|
1809
|
-
runComputeCurve,
|
|
1810
|
-
runContaminationProbe,
|
|
1811
|
-
runEvalCampaign,
|
|
1812
|
-
runRLCampaign,
|
|
1813
|
-
runwiseStepRewardSummary,
|
|
1814
|
-
selfConsistency,
|
|
1815
|
-
selfNormalizedImportanceWeighting,
|
|
1816
|
-
shuffleOrder,
|
|
1817
|
-
simFidelityReport,
|
|
1818
|
-
stepRewardsToJsonl,
|
|
1819
|
-
thompsonCurriculum,
|
|
1820
|
-
toAnthropicFormat,
|
|
1821
|
-
toDpoJsonl,
|
|
1822
|
-
toDpoRows,
|
|
1823
|
-
toGrpoJsonl,
|
|
1824
|
-
toGrpoRows,
|
|
1825
|
-
toPrmJsonl,
|
|
1826
|
-
toPrmRows,
|
|
1827
|
-
toSftJsonl,
|
|
1828
|
-
toSftRows,
|
|
1829
|
-
validateDatasetFormats,
|
|
1830
|
-
varianceBasedCurriculum,
|
|
1831
|
-
verificationReportToRunRecord
|
|
1832
|
-
};
|
|
2408
|
+
const drawMargin = input.drawMargin ?? 0;
|
|
2409
|
+
const byKey = /* @__PURE__ */ new Map();
|
|
2410
|
+
for (const r of input.runs) {
|
|
2411
|
+
const arr = byKey.get(r.matchKey) ?? [];
|
|
2412
|
+
arr.push({
|
|
2413
|
+
candidateId: r.candidateId,
|
|
2414
|
+
score: r.score
|
|
2415
|
+
});
|
|
2416
|
+
byKey.set(r.matchKey, arr);
|
|
2417
|
+
}
|
|
2418
|
+
const outcomes = [];
|
|
2419
|
+
for (const arr of byKey.values()) for (let i = 0; i < arr.length; i++) for (let j = i + 1; j < arr.length; j++) {
|
|
2420
|
+
const a = arr[i];
|
|
2421
|
+
const b = arr[j];
|
|
2422
|
+
if (a.candidateId === b.candidateId) continue;
|
|
2423
|
+
const margin = Math.abs(a.score - b.score);
|
|
2424
|
+
if (margin <= drawMargin) outcomes.push({
|
|
2425
|
+
winner: a.candidateId,
|
|
2426
|
+
loser: b.candidateId,
|
|
2427
|
+
draw: true,
|
|
2428
|
+
weight: 1
|
|
2429
|
+
});
|
|
2430
|
+
else {
|
|
2431
|
+
const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
|
|
2432
|
+
outcomes.push({
|
|
2433
|
+
winner: winner.candidateId,
|
|
2434
|
+
loser: loser.candidateId,
|
|
2435
|
+
weight: margin
|
|
2436
|
+
});
|
|
2437
|
+
}
|
|
2438
|
+
}
|
|
2439
|
+
return outcomes;
|
|
2440
|
+
}
|
|
2441
|
+
//#endregion
|
|
2442
|
+
export { ABSENT_CATEGORY, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, DPO_CONTEXT_REQUIREMENT, FileSystemOutcomeStore, InMemoryOutcomeStore, PRM_CONTEXT_REQUIREMENT, PredictiveValidityResearcher, REPRESENTATIVE_MIN_FIDELITY, STEP_REWARD_CONTEXT_REQUIREMENT, appendToCorpus, applyEloUpdate, assertPrmTrainableLine, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };
|
|
2443
|
+
|
|
1833
2444
|
//# sourceMappingURL=rl.js.map
|