@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,821 @@
|
|
|
1
|
+
//#region src/rollout/gate-checks.ts
|
|
2
|
+
/**
|
|
3
|
+
* Every gate check in the package, in the order they are applied.
|
|
4
|
+
*
|
|
5
|
+
* Order is load-bearing only for which message a caller sees first: a line that
|
|
6
|
+
* trips two checks reports the earlier one, and `reward-relationship` is first
|
|
7
|
+
* because it is the invariant the other three protect.
|
|
8
|
+
*/
|
|
9
|
+
const GATE_CHECK_IDS = [
|
|
10
|
+
"reward-relationship",
|
|
11
|
+
"gated-evidence",
|
|
12
|
+
"undeclared-step-payload",
|
|
13
|
+
"unscreened-reward"
|
|
14
|
+
];
|
|
15
|
+
/** One untyped field read off the outcome, so every check narrows from the same place. */
|
|
16
|
+
const field = (subject, name) => subject.outcome?.[name];
|
|
17
|
+
function readReward(value) {
|
|
18
|
+
if (value === null || value === void 0) return { kind: "absent" };
|
|
19
|
+
if (typeof value === "number") {
|
|
20
|
+
if (Number.isNaN(value)) return {
|
|
21
|
+
kind: "unreadable",
|
|
22
|
+
shown: "NaN"
|
|
23
|
+
};
|
|
24
|
+
return value > 0 ? {
|
|
25
|
+
kind: "positive",
|
|
26
|
+
shown: String(value)
|
|
27
|
+
} : { kind: "cleared" };
|
|
28
|
+
}
|
|
29
|
+
return {
|
|
30
|
+
kind: "unreadable",
|
|
31
|
+
shown: `${JSON.stringify(value) ?? String(value)} (${typeof value})`
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
const isPlainRecord = (value) => typeof value === "object" && value !== null && !Array.isArray(value);
|
|
35
|
+
/**
|
|
36
|
+
* Whether a reward-derived field carries anything at all.
|
|
37
|
+
*
|
|
38
|
+
* The previous emptiness test was `Object.keys(value).length > 0`, which reads
|
|
39
|
+
* `[]` for a NUMBER, a BOOLEAN and the empty string — so `metrics: 0.95` on a
|
|
40
|
+
* gated line counted as empty and shipped verbatim in the verifiers format's
|
|
41
|
+
* per-rubric score dict. `Object.keys` being total over primitives is not the
|
|
42
|
+
* same property as being CORRECT over them.
|
|
43
|
+
*
|
|
44
|
+
* The rule that has no such hole: only `undefined`, `null` and a record with no
|
|
45
|
+
* keys are empty. Everything else is payload, whatever its type.
|
|
46
|
+
*/
|
|
47
|
+
function payloadIsPopulated(value) {
|
|
48
|
+
if (value === void 0 || value === null) return false;
|
|
49
|
+
if (isPlainRecord(value)) return Object.keys(value).length > 0;
|
|
50
|
+
return true;
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Every key `tangle.rollout.v1` declares on a step, as a TOTAL map over the
|
|
54
|
+
* interface: a field added to `RolloutStep` and not to this list does not
|
|
55
|
+
* compile, so the list cannot silently fall behind the schema it mirrors.
|
|
56
|
+
*
|
|
57
|
+
* This is the producer's OWN classification, which is the only partition
|
|
58
|
+
* `gatedEvidenceOf` accepts — a key-name heuristic ("strip anything matching
|
|
59
|
+
* /reward|score/") holds until someone names a field `credit` and the next
|
|
60
|
+
* per-step signal ships at full value.
|
|
61
|
+
*/
|
|
62
|
+
const DECLARED_STEP_KEYS = {
|
|
63
|
+
kind: true,
|
|
64
|
+
name: true,
|
|
65
|
+
input: true,
|
|
66
|
+
output: true,
|
|
67
|
+
status: true,
|
|
68
|
+
durationMs: true,
|
|
69
|
+
llm_call_count: true,
|
|
70
|
+
prompt_token_ids: true,
|
|
71
|
+
completion_token_ids: true,
|
|
72
|
+
logprobs: true
|
|
73
|
+
};
|
|
74
|
+
/**
|
|
75
|
+
* The part of `steps` the wire format has no name for — a per-step reward, a
|
|
76
|
+
* per-step score, whatever a producer decided to hang there.
|
|
77
|
+
*
|
|
78
|
+
* `undefined` when the steps carry nothing undeclared.
|
|
79
|
+
*/
|
|
80
|
+
function undeclaredStepPayload(steps) {
|
|
81
|
+
if (steps === void 0 || steps === null) return void 0;
|
|
82
|
+
if (!Array.isArray(steps)) return steps;
|
|
83
|
+
const extras = [];
|
|
84
|
+
let found = false;
|
|
85
|
+
for (const step of steps) {
|
|
86
|
+
const extra = {};
|
|
87
|
+
if (isPlainRecord(step)) for (const [key, value] of Object.entries(step)) {
|
|
88
|
+
if (key in DECLARED_STEP_KEYS) continue;
|
|
89
|
+
extra[key] = value;
|
|
90
|
+
found = true;
|
|
91
|
+
}
|
|
92
|
+
else if (step !== void 0 && step !== null) {
|
|
93
|
+
extras.push({ step });
|
|
94
|
+
found = true;
|
|
95
|
+
continue;
|
|
96
|
+
}
|
|
97
|
+
extras.push(extra);
|
|
98
|
+
}
|
|
99
|
+
return found ? extras : void 0;
|
|
100
|
+
}
|
|
101
|
+
/** `steps` projected down to the keys the wire format declares. */
|
|
102
|
+
function declaredSteps(steps) {
|
|
103
|
+
if (!Array.isArray(steps)) return void 0;
|
|
104
|
+
const kept = [];
|
|
105
|
+
for (const step of steps) {
|
|
106
|
+
if (!isPlainRecord(step)) continue;
|
|
107
|
+
const projected = {};
|
|
108
|
+
for (const [key, value] of Object.entries(step)) if (key in DECLARED_STEP_KEYS) projected[key] = value;
|
|
109
|
+
kept.push(projected);
|
|
110
|
+
}
|
|
111
|
+
return kept;
|
|
112
|
+
}
|
|
113
|
+
/**
|
|
114
|
+
* THE anti-Goodhart invariant, checked as a RELATIONSHIP between two fields
|
|
115
|
+
* rather than as two independent type checks.
|
|
116
|
+
*
|
|
117
|
+
* Everything upstream of a training export is allowed to be wrong; this is the
|
|
118
|
+
* one thing that cannot be. `realness_gated: true` means the run faked its
|
|
119
|
+
* success signal, so its reward is a fabrication, and a fabrication above zero
|
|
120
|
+
* is precisely what a trainer would learn to reproduce. Validating only that
|
|
121
|
+
* `reward` is a number and `realness_gated` is a boolean is what let a line
|
|
122
|
+
* claiming `{reward: 0.95, realness_gated: true}` validate clean and walk
|
|
123
|
+
* through every exporter.
|
|
124
|
+
*/
|
|
125
|
+
const rewardRelationship = {
|
|
126
|
+
id: "reward-relationship",
|
|
127
|
+
refuses: "a positive — or unreadable — reward on a line the authenticity screen flagged as gamed",
|
|
128
|
+
tripwires: [{ outcome: {
|
|
129
|
+
reward: .95,
|
|
130
|
+
realness_gated: true
|
|
131
|
+
} }, { outcome: {
|
|
132
|
+
reward: "0.95",
|
|
133
|
+
realness_gated: true
|
|
134
|
+
} }],
|
|
135
|
+
errors: (subject) => {
|
|
136
|
+
if (field(subject, "realness_gated") !== true) return [];
|
|
137
|
+
const reading = readReward(field(subject, "reward"));
|
|
138
|
+
if (reading.kind === "absent" || reading.kind === "cleared") return [];
|
|
139
|
+
if (reading.kind === "unreadable") return [`outcome.reward: ${reading.shown} with outcome.realness_gated: true — the wire format declares this field \`number | null\`, so the gate cannot read this value as a number and cannot establish that the reward was cleared. On a run flagged as gamed that is not evidence of a zero, it is the absence of it, and a consumer that coerces the value (\`"0.95"\`, \`"2"\`) reads a positive reward off a faked success. Emit a number or null — \`trainingReward\` / \`trainingScore\` from \`rollout/reward.ts\` produce one.`];
|
|
140
|
+
return [`outcome.reward: ${reading.shown} with outcome.realness_gated: true — a run flagged as gamed may not carry a positive reward. The anti-Goodhart gate forces the reward to 0 (a real verdict: the gate decided) before the line is written, so a fine-tune cannot learn from a faked success. Derive the reward with \`trainingReward\` / \`trainingScore\` from \`rollout/reward.ts\`, or drop the line.`];
|
|
141
|
+
}
|
|
142
|
+
};
|
|
143
|
+
/**
|
|
144
|
+
* The reward-bearing outcome fields that are NOT the scalar: the numbers the
|
|
145
|
+
* reward was computed from, and the verdict record that claimed it.
|
|
146
|
+
*
|
|
147
|
+
* Returned as one block rather than filtered key-by-key. A key-name heuristic
|
|
148
|
+
* ("zero anything matching `layer.*` or `/score/`") is the same defect shape as
|
|
149
|
+
* the line-oriented regex the AST score guard replaced: it holds until someone
|
|
150
|
+
* names a metric `pass_fraction`, and the next reward-shaped key ships at full
|
|
151
|
+
* value. The producer's OWN classification — "this is the scalar, that is
|
|
152
|
+
* everything else" — is the only partition that cannot be out-guessed.
|
|
153
|
+
*/
|
|
154
|
+
function gatedEvidenceOf(subject) {
|
|
155
|
+
const evidence = {};
|
|
156
|
+
const metrics = field(subject, "metrics");
|
|
157
|
+
if (payloadIsPopulated(metrics)) evidence.metrics = metrics;
|
|
158
|
+
const verdict = field(subject, "verdict");
|
|
159
|
+
if (verdict !== null && verdict !== void 0) evidence.verdict = verdict;
|
|
160
|
+
const steps = undeclaredStepPayload(subject.steps);
|
|
161
|
+
if (steps !== void 0) evidence.steps = steps;
|
|
162
|
+
return evidence.metrics === void 0 && evidence.verdict === void 0 && steps === void 0 ? void 0 : evidence;
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* The invariant is about the OUTCOME, not about one field of it.
|
|
166
|
+
*
|
|
167
|
+
* `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`
|
|
168
|
+
* with no gate, so a gated run exported `reward: 0` (correct) while the
|
|
169
|
+
* deterministic per-layer scores that reward was COMPUTED FROM — the `layer.*`
|
|
170
|
+
* keys `rl/verifiable-reward.ts` calls the RL training signal — shipped at 1.0,
|
|
171
|
+
* in the top-level `metrics` dict of the Prime Intellect verifiers format, which
|
|
172
|
+
* IS that format's per-rubric score dict. `verdict` leaks the same way into
|
|
173
|
+
* `toRftItem`'s `reference.verdict`, where a grader author reads
|
|
174
|
+
* `resolved: true` off a run that faked it.
|
|
175
|
+
*/
|
|
176
|
+
const gatedEvidence = {
|
|
177
|
+
id: "gated-evidence",
|
|
178
|
+
refuses: "the numbers a fabricated reward was computed from, riding along at reward 0",
|
|
179
|
+
tripwires: [{ outcome: {
|
|
180
|
+
reward: 0,
|
|
181
|
+
realness_gated: true,
|
|
182
|
+
metrics: { "layer.tests": 1 }
|
|
183
|
+
} }, { outcome: {
|
|
184
|
+
reward: 0,
|
|
185
|
+
realness_gated: true,
|
|
186
|
+
metrics: .95
|
|
187
|
+
} }],
|
|
188
|
+
errors: (subject) => {
|
|
189
|
+
if (field(subject, "realness_gated") !== true) return [];
|
|
190
|
+
const evidence = gatedEvidenceOf(subject);
|
|
191
|
+
if (evidence === void 0) return [];
|
|
192
|
+
if (evidence.metrics === void 0 && evidence.verdict === void 0) return [];
|
|
193
|
+
return [`${evidence.metrics !== void 0 ? "outcome.metrics" : "outcome.verdict"} is populated with outcome.realness_gated: true — a run flagged as gamed may not ship the numbers its fabricated reward was computed from, even at reward 0. The per-layer scores in \`metrics\` ARE the reward signal in the verifiers format, and \`verdict\` is the record that claimed the success. Mint through \`assertMinted\`, which relocates both to \`provenance.gated_evidence\` (see \`gateGamedOutcome\`), or drop the line.`];
|
|
194
|
+
}
|
|
195
|
+
};
|
|
196
|
+
/**
|
|
197
|
+
* A reward-bearing field hung on `steps[]`, where the gate was not looking.
|
|
198
|
+
*
|
|
199
|
+
* The first three checks all read `outcome`, so the whole apparatus was blind to
|
|
200
|
+
* anything a producer wrote elsewhere on the line — and `toRewardRows` emits
|
|
201
|
+
* `steps` verbatim beside the scalar it just forced to 0. A generated corpus
|
|
202
|
+
* found it in twenty cases: a gated line carrying
|
|
203
|
+
* `steps: [{kind, name, reward: 0.86}]` exported `{"reward": 0, "steps":
|
|
204
|
+
* [{"reward": 0.86}]}`, which is the per-step credit assignment of a run that
|
|
205
|
+
* faked its success, at full value, through the MINTED door.
|
|
206
|
+
*
|
|
207
|
+
* `RolloutStep` declares ten keys and none of them is a reward, so the partition
|
|
208
|
+
* needs no judgement call: whatever the format does not name is unclassified
|
|
209
|
+
* payload, and on a gated line unclassified payload is exactly what the gate
|
|
210
|
+
* refuses. `assertMinted` relocates it rather than rejecting, like the sibling
|
|
211
|
+
* check, so an already-published ledger stays readable.
|
|
212
|
+
*/
|
|
213
|
+
const undeclaredStepPayloadCheck = {
|
|
214
|
+
id: "undeclared-step-payload",
|
|
215
|
+
refuses: "per-step reward signal hung on a gated line’s steps, outside every field the gate read",
|
|
216
|
+
tripwires: [{
|
|
217
|
+
outcome: {
|
|
218
|
+
reward: 0,
|
|
219
|
+
realness_gated: true
|
|
220
|
+
},
|
|
221
|
+
steps: [{
|
|
222
|
+
kind: "tool",
|
|
223
|
+
name: "edit",
|
|
224
|
+
reward: .9
|
|
225
|
+
}]
|
|
226
|
+
}],
|
|
227
|
+
errors: (subject) => {
|
|
228
|
+
if (field(subject, "realness_gated") !== true) return [];
|
|
229
|
+
const extra = undeclaredStepPayload(subject.steps);
|
|
230
|
+
if (extra === void 0) return [];
|
|
231
|
+
return [`steps[] carries fields \`${ROLLOUT_SCHEMA_NAME}\` does not declare (${JSON.stringify(extra).slice(0, 120)}) with outcome.realness_gated: true — a per-step reward is training signal exactly like the scalar, and the exporters copy \`steps\` through verbatim, so zeroing \`outcome.reward\` while leaving it in place ships the gamed run’s step-level credit assignment at full value. Mint through \`assertMinted\`, which projects the steps down to the declared keys and relocates the rest to \`provenance.gated_evidence.steps\` (see \`gateGamedOutcome\`).`];
|
|
232
|
+
}
|
|
233
|
+
};
|
|
234
|
+
/** Named here rather than imported, so `gate-checks` stays free of schema cycles. */
|
|
235
|
+
const ROLLOUT_SCHEMA_NAME = "tangle.rollout.v1";
|
|
236
|
+
/**
|
|
237
|
+
* The registry. Total over `GateCheckId`, so an id with no check does not
|
|
238
|
+
* compile, and `GATE_CHECK_IDS` stays the single enumeration everything
|
|
239
|
+
* iterates.
|
|
240
|
+
*/
|
|
241
|
+
const GATE_CHECKS = {
|
|
242
|
+
"reward-relationship": rewardRelationship,
|
|
243
|
+
"gated-evidence": gatedEvidence,
|
|
244
|
+
"undeclared-step-payload": undeclaredStepPayloadCheck,
|
|
245
|
+
"unscreened-reward": {
|
|
246
|
+
id: "unscreened-reward",
|
|
247
|
+
refuses: "a positive reward whose producer declares that no authenticity screen ever ran",
|
|
248
|
+
tripwires: [{ outcome: {
|
|
249
|
+
reward: 1,
|
|
250
|
+
realness_gated: false,
|
|
251
|
+
realness_screened: false
|
|
252
|
+
} }, { outcome: {
|
|
253
|
+
reward: "2",
|
|
254
|
+
realness_gated: false,
|
|
255
|
+
realness_screened: false
|
|
256
|
+
} }],
|
|
257
|
+
errors: (subject) => {
|
|
258
|
+
if (field(subject, "realness_screened") !== false) return [];
|
|
259
|
+
const reading = readReward(field(subject, "reward"));
|
|
260
|
+
if (reading.kind === "absent" || reading.kind === "cleared") return [];
|
|
261
|
+
return [`outcome.reward: ${reading.shown} with outcome.realness_screened: false — this producer declares that NO authenticity screen ran on this reward, so nothing has established the success is real, and an unscreened positive reward is exactly the signal the anti-Goodhart gate exists to qualify. Screen the run and write the verdict (\`rolloutRewardFields\` from a \`RunRecord\` carrying \`outcome.realness\`), or emit \`reward: null\` — an unqualified verdict is a labeled gap, not a measured success.`];
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
};
|
|
265
|
+
const enforced = { kind: "enforce" };
|
|
266
|
+
const repairedBy = (by) => ({
|
|
267
|
+
kind: "repair",
|
|
268
|
+
by
|
|
269
|
+
});
|
|
270
|
+
const omittedBecause = (because) => ({
|
|
271
|
+
kind: "omit",
|
|
272
|
+
because
|
|
273
|
+
});
|
|
274
|
+
/**
|
|
275
|
+
* Every entry point that decides about the gate, and what it decides.
|
|
276
|
+
*
|
|
277
|
+
* Read this as the package's gate policy in one screen. The four entry points
|
|
278
|
+
* are not interchangeable — a validator that rejects, a mint funnel that
|
|
279
|
+
* repairs, a runtime backstop for untyped callers, and a release certifier over
|
|
280
|
+
* emitted rows — and the dispositions say which is which.
|
|
281
|
+
*/
|
|
282
|
+
const GATE_POLICIES = {
|
|
283
|
+
/**
|
|
284
|
+
* The schema validator. Rejects the reward relationship and NOTHING ELSE, on
|
|
285
|
+
* purpose: it runs on every line read off disk, and the other two conditions
|
|
286
|
+
* describe artifacts that already exist.
|
|
287
|
+
*/
|
|
288
|
+
validateRolloutLine: {
|
|
289
|
+
"reward-relationship": enforced,
|
|
290
|
+
"gated-evidence": omittedBecause("a ledger written before the relocation existed carries `metrics` on its gated lines; rejecting those would make every already-published artifact unreadable. The condition is REPAIRED at `assertMinted` (`gateGamedOutcome`) instead, which is the only door into the training path, so the artifact stays readable and the leak still closes."),
|
|
291
|
+
"undeclared-step-payload": omittedBecause("identical reasoning to `gated-evidence`, one field over: a foreign or pre-unification ledger may carry producer-invented keys on its steps, and refusing to READ those files buys nothing the relocation at `assertMinted` does not already buy. Promotion into training is where it closes."),
|
|
292
|
+
"unscreened-reward": omittedBecause("supervision-journal rows legitimately carry an unscreened positive reward (there is no `RunRecord.outcome.realness` behind them) and must stay writable, readable and reportable. Only PROMOTION into training is closed, at `assertMinted`.")
|
|
293
|
+
},
|
|
294
|
+
/**
|
|
295
|
+
* The mint funnel — the single door every `MintedRolloutLine` passes. Validates
|
|
296
|
+
* first (so the reward relationship has already been rejected), then refuses
|
|
297
|
+
* what cannot be repaired, then repairs what can.
|
|
298
|
+
*/
|
|
299
|
+
assertMinted: {
|
|
300
|
+
"reward-relationship": enforced,
|
|
301
|
+
"gated-evidence": repairedBy("gateGamedOutcome"),
|
|
302
|
+
"undeclared-step-payload": repairedBy("gateGamedOutcome"),
|
|
303
|
+
"unscreened-reward": enforced
|
|
304
|
+
},
|
|
305
|
+
/**
|
|
306
|
+
* The runtime backstop, and the one entry point with no license to omit
|
|
307
|
+
* anything: it exists for callers the type system never saw (plain JavaScript
|
|
308
|
+
* handing an object literal to a published exporter), so a check it skips is a
|
|
309
|
+
* check that does not run at all for them. This is where the fourth leak was.
|
|
310
|
+
*/
|
|
311
|
+
assertRewardGate: {
|
|
312
|
+
"reward-relationship": enforced,
|
|
313
|
+
"gated-evidence": enforced,
|
|
314
|
+
"undeclared-step-payload": enforced,
|
|
315
|
+
"unscreened-reward": enforced
|
|
316
|
+
},
|
|
317
|
+
/**
|
|
318
|
+
* The release certifier. Same checks, measured over the rows a release is
|
|
319
|
+
* ABOUT TO WRITE rather than over one line's outcome — see `REPORT_MEASURES`
|
|
320
|
+
* in `release/gate-report.ts`, which is the second total map this policy
|
|
321
|
+
* drives.
|
|
322
|
+
*/
|
|
323
|
+
assertGateReport: {
|
|
324
|
+
"reward-relationship": enforced,
|
|
325
|
+
"gated-evidence": enforced,
|
|
326
|
+
"undeclared-step-payload": enforced,
|
|
327
|
+
"unscreened-reward": enforced
|
|
328
|
+
}
|
|
329
|
+
};
|
|
330
|
+
/**
|
|
331
|
+
* Run the checks one entry point enforces. The ONLY way an entry point should
|
|
332
|
+
* obtain gate errors — hand-composing two of the three is the bug this module
|
|
333
|
+
* exists to remove.
|
|
334
|
+
*/
|
|
335
|
+
function gateErrors(subject, policy) {
|
|
336
|
+
const errors = [];
|
|
337
|
+
for (const id of GATE_CHECK_IDS) {
|
|
338
|
+
if (policy[id].kind !== "enforce") continue;
|
|
339
|
+
errors.push(...GATE_CHECKS[id].errors(subject));
|
|
340
|
+
}
|
|
341
|
+
return errors;
|
|
342
|
+
}
|
|
343
|
+
//#endregion
|
|
344
|
+
//#region src/rollout/schema.ts
|
|
345
|
+
/**
|
|
346
|
+
* `tangle.rollout.v1` — THE canonical rollout serialization, owned by
|
|
347
|
+
* agent-eval. One JSONL line per agent invocation (a solo eval run, a
|
|
348
|
+
* supervisor episode, a worker session, a proposer shot, a judge call, an
|
|
349
|
+
* analyst pass), labeled with its task/split coordinates and a single
|
|
350
|
+
* scalar reward, carrying the FULL message transcript inline.
|
|
351
|
+
*
|
|
352
|
+
* This schema is the reconciliation of two prior producers:
|
|
353
|
+
* - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
|
|
354
|
+
* provenance hashes, the realness gate travelling into the reward,
|
|
355
|
+
* trace-derived steps.
|
|
356
|
+
* - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
|
|
357
|
+
* role, task.split/rep, parent_rollout_id, policy provenance, capture
|
|
358
|
+
* provenance, inline canonical chat-with-tools messages.
|
|
359
|
+
* Where the two conflicted, RunRecord-derived semantics won; the wire
|
|
360
|
+
* field names follow the ledger (snake_case). See `docs/rollout.md` for
|
|
361
|
+
* the field-by-field decision table.
|
|
362
|
+
*
|
|
363
|
+
* Messages are inlined — never referenced — because every harness store a
|
|
364
|
+
* rollout can be recovered from is mutable or garbage-collected. A line
|
|
365
|
+
* must stay a complete training/eval example on its own.
|
|
366
|
+
*
|
|
367
|
+
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
368
|
+
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
369
|
+
* flag: a gated line must never export as a positive training example.
|
|
370
|
+
*
|
|
371
|
+
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
372
|
+
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
373
|
+
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
374
|
+
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
375
|
+
* training export. The relationship between the two IS the invariant, so it is
|
|
376
|
+
* checked where every other structural claim about a line is checked.
|
|
377
|
+
*
|
|
378
|
+
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
379
|
+
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
380
|
+
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
381
|
+
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
382
|
+
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
383
|
+
* minted line passes — and the reward-bearing components are relocated to
|
|
384
|
+
* `provenance.gated_evidence`, which no exporter projects.
|
|
385
|
+
*
|
|
386
|
+
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
387
|
+
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
388
|
+
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
389
|
+
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
390
|
+
* here without anyone editing this file, and a check deliberately skipped has to
|
|
391
|
+
* name itself there.
|
|
392
|
+
*/
|
|
393
|
+
const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
394
|
+
const ROLLOUT_ROLES = [
|
|
395
|
+
"agent",
|
|
396
|
+
"supervisor",
|
|
397
|
+
"worker",
|
|
398
|
+
"proposer",
|
|
399
|
+
"judge",
|
|
400
|
+
"analyst"
|
|
401
|
+
];
|
|
402
|
+
const ROLLOUT_SPLITS = [
|
|
403
|
+
"search",
|
|
404
|
+
"dev",
|
|
405
|
+
"holdout",
|
|
406
|
+
"canary"
|
|
407
|
+
];
|
|
408
|
+
/** Splits that may ship in training exports. Everything else is fail-closed excluded. */
|
|
409
|
+
const TRAINABLE_SPLITS = ["search"];
|
|
410
|
+
function isTrainableSplit(split) {
|
|
411
|
+
return TRAINABLE_SPLITS.includes(split);
|
|
412
|
+
}
|
|
413
|
+
const ROLLOUT_CAPTURES = [
|
|
414
|
+
"mint",
|
|
415
|
+
"settle-time",
|
|
416
|
+
"backfill"
|
|
417
|
+
];
|
|
418
|
+
const CHAT_ROLES = [
|
|
419
|
+
"system",
|
|
420
|
+
"user",
|
|
421
|
+
"assistant",
|
|
422
|
+
"tool"
|
|
423
|
+
];
|
|
424
|
+
const isRecord = (v) => typeof v === "object" && v !== null && !Array.isArray(v);
|
|
425
|
+
const isNumberOrNull = (v) => v === null || typeof v === "number";
|
|
426
|
+
const isNumberArray = (v) => Array.isArray(v) && v.every((n) => typeof n === "number" && Number.isFinite(n));
|
|
427
|
+
const isIntegerArray = (v) => Array.isArray(v) && v.every((n) => Number.isInteger(n));
|
|
428
|
+
const isStringOrNull = (v) => v === null || typeof v === "string";
|
|
429
|
+
const isIntegerOrNull = (v) => v === null || Number.isInteger(v);
|
|
430
|
+
function validateChatMessage(value, path, errors) {
|
|
431
|
+
if (!isRecord(value)) {
|
|
432
|
+
errors.push(`${path}: not an object`);
|
|
433
|
+
return;
|
|
434
|
+
}
|
|
435
|
+
if (!CHAT_ROLES.includes(value.role)) errors.push(`${path}.role: invalid role ${String(value.role)}`);
|
|
436
|
+
if (!isStringOrNull(value.content)) errors.push(`${path}.content: must be string|null`);
|
|
437
|
+
if (value.reasoning_content !== void 0 && typeof value.reasoning_content !== "string") errors.push(`${path}.reasoning_content: must be string when present`);
|
|
438
|
+
if (value.tool_call_id !== void 0 && typeof value.tool_call_id !== "string") errors.push(`${path}.tool_call_id: must be string when present`);
|
|
439
|
+
if (value.role === "tool" && typeof value.tool_call_id !== "string") errors.push(`${path}.tool_call_id: required on role:"tool"`);
|
|
440
|
+
if (value.is_copied_context !== void 0 && typeof value.is_copied_context !== "boolean") errors.push(`${path}.is_copied_context: must be boolean when present`);
|
|
441
|
+
if (value.tool_calls !== void 0) if (!Array.isArray(value.tool_calls)) errors.push(`${path}.tool_calls: must be an array when present`);
|
|
442
|
+
else value.tool_calls.forEach((call, i) => {
|
|
443
|
+
if (!isRecord(call) || typeof call.id !== "string" || call.type !== "function") {
|
|
444
|
+
errors.push(`${path}.tool_calls[${i}]: must be {id, type:"function", function}`);
|
|
445
|
+
return;
|
|
446
|
+
}
|
|
447
|
+
const fn = call.function;
|
|
448
|
+
if (!isRecord(fn) || typeof fn.name !== "string" || typeof fn.arguments !== "string") errors.push(`${path}.tool_calls[${i}].function: must be {name: string, arguments: string}`);
|
|
449
|
+
});
|
|
450
|
+
}
|
|
451
|
+
function validateSection(value, path, fields, errors) {
|
|
452
|
+
if (!isRecord(value)) {
|
|
453
|
+
errors.push(`${path}: not an object`);
|
|
454
|
+
return;
|
|
455
|
+
}
|
|
456
|
+
for (const [name, check, expect] of fields) if (!check(value[name])) errors.push(`${path}.${name}: expected ${expect}`);
|
|
457
|
+
}
|
|
458
|
+
/**
|
|
459
|
+
* THE anti-Goodhart gate, applied to the WHOLE outcome as a TRANSFORMATION.
|
|
460
|
+
*
|
|
461
|
+
* Two prior rounds enforced the gate as a CHECK ON ONE FIELD at N call sites,
|
|
462
|
+
* and each round the next reward-bearing field leaked. The one that shipped:
|
|
463
|
+
* `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`
|
|
464
|
+
* with no gate, so a gated run exported `reward: 0` (correct) while the
|
|
465
|
+
* deterministic per-layer scores that reward was COMPUTED FROM — the
|
|
466
|
+
* `layer.*` keys `rl/verifiable-reward.ts` calls the RL training signal —
|
|
467
|
+
* shipped at 1.0, in the top-level `metrics` dict of the Prime Intellect
|
|
468
|
+
* verifiers format, which IS that format's per-rubric score dict. `verdict`
|
|
469
|
+
* leaks the same way into `toRftItem`'s `reference.verdict`, where a grader
|
|
470
|
+
* author reads `resolved: true` off a run that faked it.
|
|
471
|
+
*
|
|
472
|
+
* So the rule is no longer "zero the field we remembered". It is: if the gate
|
|
473
|
+
* fired, the outcome that leaves here carries NOTHING positive that was derived
|
|
474
|
+
* from the reward, whichever field a present or future exporter decides to
|
|
475
|
+
* read. `reward` is already forced to 0 upstream (`trainingReward`) and
|
|
476
|
+
* REJECTED here if it is not; `metrics` and `verdict` are relocated.
|
|
477
|
+
*
|
|
478
|
+
* WHERE they go, and why relocation rather than deletion: zeroing destroys the
|
|
479
|
+
* audit trail that shows why the run was gated and what it claimed, which is
|
|
480
|
+
* the row an auditor most wants and the labeled example a gaming DETECTOR
|
|
481
|
+
* trains on. `provenance.gated_evidence` keeps every byte, at a path no
|
|
482
|
+
* training exporter reads — all four release configs and every `rl/exporters`
|
|
483
|
+
* shape project from `outcome`, `messages`, `cost` and `task`; none projects
|
|
484
|
+
* `provenance`. Auditability preserved, training signal removed, and a future
|
|
485
|
+
* exporter that reads a field nobody thought of is safe by construction because
|
|
486
|
+
* the field is empty rather than because the exporter remembered to check.
|
|
487
|
+
*
|
|
488
|
+
* Idempotent: a second application finds nothing left to move and returns the
|
|
489
|
+
* line unchanged, so re-minting a line read back off a ledger cannot clobber
|
|
490
|
+
* the evidence it already carries.
|
|
491
|
+
*/
|
|
492
|
+
function gateGamedOutcome(line) {
|
|
493
|
+
if (line.outcome.realness_gated !== true) return line;
|
|
494
|
+
const moved = gatedEvidenceOf(line);
|
|
495
|
+
if (moved === void 0) return line;
|
|
496
|
+
const kept = line.provenance.gated_evidence;
|
|
497
|
+
const evidence = {};
|
|
498
|
+
const metrics = {
|
|
499
|
+
...kept?.metrics,
|
|
500
|
+
...moved.metrics
|
|
501
|
+
};
|
|
502
|
+
if (Object.keys(metrics).length > 0) evidence.metrics = metrics;
|
|
503
|
+
if (moved.verdict !== void 0) evidence.verdict = moved.verdict;
|
|
504
|
+
else if (kept?.verdict !== void 0) evidence.verdict = kept.verdict;
|
|
505
|
+
if (moved.steps !== void 0) evidence.steps = moved.steps;
|
|
506
|
+
else if (kept?.steps !== void 0) evidence.steps = kept.steps;
|
|
507
|
+
const steps = moved.steps === void 0 ? line.steps : declaredSteps(line.steps);
|
|
508
|
+
return {
|
|
509
|
+
...line,
|
|
510
|
+
...steps === void 0 ? {} : { steps },
|
|
511
|
+
outcome: {
|
|
512
|
+
...line.outcome,
|
|
513
|
+
metrics: {},
|
|
514
|
+
verdict: null
|
|
515
|
+
},
|
|
516
|
+
provenance: {
|
|
517
|
+
...line.provenance,
|
|
518
|
+
gated_evidence: evidence
|
|
519
|
+
}
|
|
520
|
+
};
|
|
521
|
+
}
|
|
522
|
+
/**
|
|
523
|
+
* EVERY gate check, for the export path — the runtime backstop.
|
|
524
|
+
*
|
|
525
|
+
* `validateRolloutLine` checks the reward relationship too, but an exporter
|
|
526
|
+
* cannot afford to re-validate every field of every line, and more importantly
|
|
527
|
+
* it is not the exporter's job to re-check the schema — it is its job never to
|
|
528
|
+
* emit a signal it was told is fabricated. Thrown rather than filtered: an
|
|
529
|
+
* exporter silently dropping a poisoned line would hide the producer that made
|
|
530
|
+
* it, and the producer is the actual defect.
|
|
531
|
+
*
|
|
532
|
+
* This is the third layer, and it exists for exactly one caller: JavaScript.
|
|
533
|
+
* The brand stops TypeScript callers at compile time and `assertMinted` fixes
|
|
534
|
+
* data arriving from disk, but neither is present for a plain JS consumer of
|
|
535
|
+
* the published package handing an object literal to `toRewardRows`. Which is
|
|
536
|
+
* exactly why this entry point's policy omits nothing: a check it skips is a
|
|
537
|
+
* check that never runs for those callers at all. It composed two of the three
|
|
538
|
+
* for one round, and a never-screened positive reward walked through all four
|
|
539
|
+
* waist exporters at full value. `GATE_POLICIES.assertRewardGate` is now the
|
|
540
|
+
* only place that list is written down.
|
|
541
|
+
*/
|
|
542
|
+
function assertRewardGate(line, context) {
|
|
543
|
+
const errors = gateErrors(line, GATE_POLICIES.assertRewardGate);
|
|
544
|
+
if (errors.length > 0) throw new Error(`${context}: rollout ${line.rollout_id} — ${errors[0]}`);
|
|
545
|
+
}
|
|
546
|
+
function validateRolloutLine(value) {
|
|
547
|
+
const errors = [];
|
|
548
|
+
if (!isRecord(value)) return ["line: not an object"];
|
|
549
|
+
if (value.schema !== "tangle.rollout.v1") errors.push(`schema: expected "${ROLLOUT_SCHEMA}"`);
|
|
550
|
+
if (typeof value.rollout_id !== "string" || value.rollout_id.length === 0) errors.push("rollout_id: expected non-empty string");
|
|
551
|
+
if (!isStringOrNull(value.parent_rollout_id)) errors.push("parent_rollout_id: expected string|null");
|
|
552
|
+
if (typeof value.run_id !== "string" || value.run_id.length === 0) errors.push("run_id: expected non-empty string");
|
|
553
|
+
if (!isStringOrNull(value.experiment_id)) errors.push("experiment_id: expected string|null");
|
|
554
|
+
if (!isStringOrNull(value.candidate_id)) errors.push("candidate_id: expected string|null");
|
|
555
|
+
if (!isIntegerOrNull(value.generation)) errors.push("generation: expected integer|null");
|
|
556
|
+
if (!isIntegerOrNull(value.candidate_index)) errors.push("candidate_index: expected integer|null");
|
|
557
|
+
if (!ROLLOUT_ROLES.includes(value.role)) errors.push(`role: invalid role ${String(value.role)}`);
|
|
558
|
+
validateSection(value.task, "task", [
|
|
559
|
+
[
|
|
560
|
+
"suite",
|
|
561
|
+
(v) => typeof v === "string" && v.length > 0,
|
|
562
|
+
"non-empty string"
|
|
563
|
+
],
|
|
564
|
+
[
|
|
565
|
+
"instance_id",
|
|
566
|
+
(v) => typeof v === "string" && v.length > 0,
|
|
567
|
+
"non-empty string"
|
|
568
|
+
],
|
|
569
|
+
[
|
|
570
|
+
"split",
|
|
571
|
+
(v) => ROLLOUT_SPLITS.includes(v),
|
|
572
|
+
`one of ${ROLLOUT_SPLITS.join("|")}`
|
|
573
|
+
],
|
|
574
|
+
[
|
|
575
|
+
"seed",
|
|
576
|
+
isNumberOrNull,
|
|
577
|
+
"number|null"
|
|
578
|
+
],
|
|
579
|
+
[
|
|
580
|
+
"rep",
|
|
581
|
+
(v) => Number.isInteger(v),
|
|
582
|
+
"integer"
|
|
583
|
+
]
|
|
584
|
+
], errors);
|
|
585
|
+
validateSection(value.policy, "policy", [
|
|
586
|
+
[
|
|
587
|
+
"harness",
|
|
588
|
+
isStringOrNull,
|
|
589
|
+
"string|null"
|
|
590
|
+
],
|
|
591
|
+
[
|
|
592
|
+
"harness_version",
|
|
593
|
+
isStringOrNull,
|
|
594
|
+
"string|null"
|
|
595
|
+
],
|
|
596
|
+
[
|
|
597
|
+
"model",
|
|
598
|
+
isStringOrNull,
|
|
599
|
+
"string|null"
|
|
600
|
+
],
|
|
601
|
+
[
|
|
602
|
+
"provider",
|
|
603
|
+
isStringOrNull,
|
|
604
|
+
"string|null"
|
|
605
|
+
],
|
|
606
|
+
[
|
|
607
|
+
"profile_commit",
|
|
608
|
+
isStringOrNull,
|
|
609
|
+
"string|null"
|
|
610
|
+
],
|
|
611
|
+
[
|
|
612
|
+
"sampling",
|
|
613
|
+
(v) => v === null || isRecord(v),
|
|
614
|
+
"object|null"
|
|
615
|
+
]
|
|
616
|
+
], errors);
|
|
617
|
+
if (isRecord(value.policy)) {
|
|
618
|
+
for (const key of [
|
|
619
|
+
"prompt_hash",
|
|
620
|
+
"config_hash",
|
|
621
|
+
"agent_profile_cell_id"
|
|
622
|
+
]) if (value.policy[key] !== void 0 && !isStringOrNull(value.policy[key])) errors.push(`policy.${key}: expected string|null when present`);
|
|
623
|
+
}
|
|
624
|
+
if (!Array.isArray(value.messages)) errors.push("messages: expected array");
|
|
625
|
+
else for (const [i, m] of value.messages.entries()) validateChatMessage(m, `messages[${i}]`, errors);
|
|
626
|
+
if (!Array.isArray(value.tool_defs)) errors.push("tool_defs: expected array");
|
|
627
|
+
else value.tool_defs.forEach((d, i) => {
|
|
628
|
+
if (!isRecord(d) || d.type !== "function" || !isRecord(d.function) || typeof d.function.name !== "string") errors.push(`tool_defs[${i}]: must be {type:"function", function:{name}}`);
|
|
629
|
+
});
|
|
630
|
+
if (value.steps !== void 0) if (!Array.isArray(value.steps)) errors.push("steps: expected array when present");
|
|
631
|
+
else value.steps.forEach((s, i) => {
|
|
632
|
+
if (!isRecord(s) || typeof s.kind !== "string" || typeof s.name !== "string") {
|
|
633
|
+
errors.push(`steps[${i}]: must be {kind: string, name: string, …}`);
|
|
634
|
+
return;
|
|
635
|
+
}
|
|
636
|
+
if (s.llm_call_count !== void 0 && !Number.isInteger(s.llm_call_count)) errors.push(`steps[${i}].llm_call_count: expected integer when present`);
|
|
637
|
+
for (const key of ["prompt_token_ids", "completion_token_ids"]) if (s[key] !== void 0 && !isIntegerArray(s[key])) errors.push(`steps[${i}].${key}: expected integer[] when present`);
|
|
638
|
+
if (s.logprobs !== void 0 && !isNumberArray(s.logprobs)) errors.push(`steps[${i}].logprobs: expected number[] when present`);
|
|
639
|
+
});
|
|
640
|
+
validateSection(value.outcome, "outcome", [
|
|
641
|
+
[
|
|
642
|
+
"reward",
|
|
643
|
+
isNumberOrNull,
|
|
644
|
+
"number|null"
|
|
645
|
+
],
|
|
646
|
+
[
|
|
647
|
+
"reward_source",
|
|
648
|
+
isStringOrNull,
|
|
649
|
+
"string|null"
|
|
650
|
+
],
|
|
651
|
+
[
|
|
652
|
+
"metrics",
|
|
653
|
+
isRecord,
|
|
654
|
+
"object"
|
|
655
|
+
],
|
|
656
|
+
[
|
|
657
|
+
"is_completed",
|
|
658
|
+
(v) => typeof v === "boolean",
|
|
659
|
+
"boolean"
|
|
660
|
+
],
|
|
661
|
+
[
|
|
662
|
+
"is_truncated",
|
|
663
|
+
(v) => typeof v === "boolean",
|
|
664
|
+
"boolean"
|
|
665
|
+
],
|
|
666
|
+
[
|
|
667
|
+
"error",
|
|
668
|
+
isStringOrNull,
|
|
669
|
+
"string|null"
|
|
670
|
+
]
|
|
671
|
+
], errors);
|
|
672
|
+
if (isRecord(value.outcome)) {
|
|
673
|
+
if (!("verdict" in value.outcome)) errors.push("outcome.verdict: field required (may be null)");
|
|
674
|
+
if (typeof value.outcome.realness_gated !== "boolean") errors.push("outcome.realness_gated: expected boolean");
|
|
675
|
+
if (value.outcome.realness_screened !== void 0 && typeof value.outcome.realness_screened !== "boolean") errors.push("outcome.realness_screened: expected boolean when present");
|
|
676
|
+
errors.push(...gateErrors({
|
|
677
|
+
outcome: value.outcome,
|
|
678
|
+
steps: value.steps
|
|
679
|
+
}, GATE_POLICIES.validateRolloutLine));
|
|
680
|
+
}
|
|
681
|
+
validateSection(value.cost, "cost", [
|
|
682
|
+
[
|
|
683
|
+
"usd",
|
|
684
|
+
isNumberOrNull,
|
|
685
|
+
"number|null"
|
|
686
|
+
],
|
|
687
|
+
[
|
|
688
|
+
"tokens_in",
|
|
689
|
+
isNumberOrNull,
|
|
690
|
+
"number|null"
|
|
691
|
+
],
|
|
692
|
+
[
|
|
693
|
+
"tokens_out",
|
|
694
|
+
isNumberOrNull,
|
|
695
|
+
"number|null"
|
|
696
|
+
],
|
|
697
|
+
[
|
|
698
|
+
"tokens_reasoning",
|
|
699
|
+
isNumberOrNull,
|
|
700
|
+
"number|null"
|
|
701
|
+
],
|
|
702
|
+
[
|
|
703
|
+
"cache_read",
|
|
704
|
+
isNumberOrNull,
|
|
705
|
+
"number|null"
|
|
706
|
+
],
|
|
707
|
+
[
|
|
708
|
+
"cache_write",
|
|
709
|
+
isNumberOrNull,
|
|
710
|
+
"number|null"
|
|
711
|
+
],
|
|
712
|
+
[
|
|
713
|
+
"wall_s",
|
|
714
|
+
isNumberOrNull,
|
|
715
|
+
"number|null"
|
|
716
|
+
]
|
|
717
|
+
], errors);
|
|
718
|
+
if (isRecord(value.cost) && value.cost.llm_call_count !== void 0 && !(value.cost.llm_call_count === null || Number.isInteger(value.cost.llm_call_count))) errors.push("cost.llm_call_count: expected integer|null when present");
|
|
719
|
+
validateSection(value.artifacts, "artifacts", [
|
|
720
|
+
[
|
|
721
|
+
"patch_path",
|
|
722
|
+
isStringOrNull,
|
|
723
|
+
"string|null"
|
|
724
|
+
],
|
|
725
|
+
[
|
|
726
|
+
"run_dir",
|
|
727
|
+
isStringOrNull,
|
|
728
|
+
"string|null"
|
|
729
|
+
],
|
|
730
|
+
[
|
|
731
|
+
"transcript_ref",
|
|
732
|
+
isStringOrNull,
|
|
733
|
+
"string|null"
|
|
734
|
+
]
|
|
735
|
+
], errors);
|
|
736
|
+
validateSection(value.provenance, "provenance", [[
|
|
737
|
+
"captured_at",
|
|
738
|
+
(v) => typeof v === "string" && !Number.isNaN(Date.parse(v)),
|
|
739
|
+
"ISO-8601 timestamp"
|
|
740
|
+
], [
|
|
741
|
+
"capture",
|
|
742
|
+
(v) => ROLLOUT_CAPTURES.includes(v),
|
|
743
|
+
`one of ${ROLLOUT_CAPTURES.join("|")}`
|
|
744
|
+
]], errors);
|
|
745
|
+
if (isRecord(value.provenance)) {
|
|
746
|
+
if (value.provenance.gap !== void 0 && typeof value.provenance.gap !== "string") errors.push("provenance.gap: must be string when present");
|
|
747
|
+
const evidence = value.provenance.gated_evidence;
|
|
748
|
+
if (evidence !== void 0) {
|
|
749
|
+
if (!isRecord(evidence)) errors.push("provenance.gated_evidence: must be an object when present");
|
|
750
|
+
else if (evidence.metrics !== void 0 && !isRecord(evidence.metrics)) errors.push("provenance.gated_evidence.metrics: must be an object when present");
|
|
751
|
+
}
|
|
752
|
+
}
|
|
753
|
+
if (Array.isArray(value.messages) && isRecord(value.provenance)) {
|
|
754
|
+
if (value.messages.length === 0 && typeof value.provenance.gap !== "string") errors.push("provenance.gap: required when messages is empty");
|
|
755
|
+
}
|
|
756
|
+
return errors;
|
|
757
|
+
}
|
|
758
|
+
function assertRolloutLine(value, context = "rollout line") {
|
|
759
|
+
const errors = validateRolloutLine(value);
|
|
760
|
+
if (errors.length > 0) throw new Error(`invalid ${context}:\n ${errors.join("\n ")}`);
|
|
761
|
+
}
|
|
762
|
+
function isRolloutLine(value) {
|
|
763
|
+
return validateRolloutLine(value).length === 0;
|
|
764
|
+
}
|
|
765
|
+
/**
|
|
766
|
+
* Promote a line to the type the training exporters accept, applying the
|
|
767
|
+
* anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch
|
|
768
|
+
* — grep `assertMinted` to enumerate every place a line enters the training
|
|
769
|
+
* path without coming from mint or a ledger.
|
|
770
|
+
*
|
|
771
|
+
* The gate runs HERE, once, rather than at each producer, because this is the
|
|
772
|
+
* single funnel every minted line passes: `mintRolloutRows` calls it,
|
|
773
|
+
* `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the
|
|
774
|
+
* way out of a release, and a hand-built line has no other door. One
|
|
775
|
+
* transformation at the funnel means an already-published ledger holding a
|
|
776
|
+
* gated line with populated `metrics` is RE-GATED when it is read, instead of
|
|
777
|
+
* being rejected (which would make every such artifact unreadable) or trusted
|
|
778
|
+
* (which is the leak). Three steps, in this order:
|
|
779
|
+
*
|
|
780
|
+
* 1. VALIDATE the schema.
|
|
781
|
+
* 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today
|
|
782
|
+
* the reward relationship (which stays a REJECTION: a caller claiming
|
|
783
|
+
* `{reward: 0.95, realness_gated: true}` is a producer defect and must fail
|
|
784
|
+
* loudly, since laundering it into `reward: 0` here would hide the
|
|
785
|
+
* producer) and a positive reward the producer declared it never screened.
|
|
786
|
+
* 3. TRANSFORM the one check that policy marks `repair` — relocate the
|
|
787
|
+
* reward's components off `outcome` (`gateGamedOutcome`), so no exporter
|
|
788
|
+
* can leak them whichever field it reads.
|
|
789
|
+
*
|
|
790
|
+
* Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced
|
|
791
|
+
* here the moment its disposition in that policy says so.
|
|
792
|
+
*
|
|
793
|
+
* Also normalizes the optional wire flag to an explicit boolean.
|
|
794
|
+
* `realness_gated` is absent on pre-unification ledgers and absent means "not
|
|
795
|
+
* flagged" per the schema, so filling it in states a claim the line was already
|
|
796
|
+
* making, and makes the flag readable on every published row instead of most of
|
|
797
|
+
* them. `realness_screened` is NOT filled in: absent means "unknown", and
|
|
798
|
+
* inventing either value there would be the same overclaim this round removed.
|
|
799
|
+
*/
|
|
800
|
+
function assertMinted(value, context = "rollout line") {
|
|
801
|
+
assertRolloutLine(value, context);
|
|
802
|
+
const refused = gateErrors(value, GATE_POLICIES.assertMinted);
|
|
803
|
+
if (refused.length > 0) throw new Error(`invalid ${context}: rollout ${value.rollout_id} — ${refused[0]}`);
|
|
804
|
+
const gated = gateGamedOutcome(value);
|
|
805
|
+
if (gated.outcome.realness_gated === void 0) return {
|
|
806
|
+
...gated,
|
|
807
|
+
outcome: {
|
|
808
|
+
...gated.outcome,
|
|
809
|
+
realness_gated: false
|
|
810
|
+
}
|
|
811
|
+
};
|
|
812
|
+
return gated;
|
|
813
|
+
}
|
|
814
|
+
/** `assertMinted` over a batch, naming the offending index in the error. */
|
|
815
|
+
function assertMintedLines(values, context = "rollout line") {
|
|
816
|
+
return values.map((value, i) => assertMinted(value, `${context} [${i}]`));
|
|
817
|
+
}
|
|
818
|
+
//#endregion
|
|
819
|
+
export { GATE_POLICIES as _, ROLLOUT_SPLITS as a, undeclaredStepPayload as b, assertMintedLines as c, gateGamedOutcome as d, isRolloutLine as f, GATE_CHECK_IDS as g, GATE_CHECKS as h, ROLLOUT_SCHEMA as i, assertRewardGate as l, validateRolloutLine as m, ROLLOUT_CAPTURES as n, TRAINABLE_SPLITS as o, isTrainableSplit as p, ROLLOUT_ROLES as r, assertMinted as s, CHAT_ROLES as t, assertRolloutLine as u, gateErrors as v, gatedEvidenceOf as y };
|
|
820
|
+
|
|
821
|
+
//# sourceMappingURL=schema-C6DW4ZHR.js.map
|