@tangle-network/agent-eval 0.129.0 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,926 @@
|
|
|
1
|
+
import { a as RunRecord } from "./run-record-CnZu_gjl.js";
|
|
2
|
+
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
3
|
+
import { a as GatedEvidence, b as RolloutSplit, g as RolloutOutcome, h as RolloutLine, n as ChatMessage, o as MintedRolloutLine, w as ToolDef, x as RolloutStep, y as RolloutRole } from "./schema-Cef2cFmb.js";
|
|
4
|
+
import { DatabaseSync } from "node:sqlite";
|
|
5
|
+
//#region src/rollout/exporters.d.ts
|
|
6
|
+
/**
|
|
7
|
+
* The gate's two claims, which travel TOGETHER on every emitted row.
|
|
8
|
+
*
|
|
9
|
+
* `realness_gated` alone is ambiguous, and the ambiguity is exploitable:
|
|
10
|
+
* `false` reads as "we screened it and nothing fired", so a producer that has no
|
|
11
|
+
* screen at all emitted rows indistinguishable from screened-clean ones, and
|
|
12
|
+
* every consumer of the published dataset read them as clean. The second field
|
|
13
|
+
* is what separates the two claims, and it only removes the ambiguity if it
|
|
14
|
+
* reaches the WIRE — for a round it existed on `RolloutOutcome` and on no
|
|
15
|
+
* exported row shape at all, which left the published rows exactly as ambiguous
|
|
16
|
+
* as before.
|
|
17
|
+
*
|
|
18
|
+
* So there is one helper and every row shape spreads it. A row that states one
|
|
19
|
+
* claim without the other is not constructible by copying the pattern, and
|
|
20
|
+
* `exporters.test.ts` walks every emitted shape to prove none does.
|
|
21
|
+
*/
|
|
22
|
+
interface RealnessLabels {
|
|
23
|
+
/** The screen's VERDICT: the run faked its success signal. */
|
|
24
|
+
realness_gated: boolean;
|
|
25
|
+
/**
|
|
26
|
+
* Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its
|
|
27
|
+
* verdict. `false` = the producer declares it has none. `null` = not stated
|
|
28
|
+
* (pre-unification producers), which is "unknown" and never "clean".
|
|
29
|
+
*/
|
|
30
|
+
realness_screened: boolean | null;
|
|
31
|
+
}
|
|
32
|
+
declare function realnessLabels(line: MintedRolloutLine): RealnessLabels;
|
|
33
|
+
interface TrainingExportOptions {
|
|
34
|
+
/** Include held-out evaluation data in training output. Default false. */
|
|
35
|
+
allowHeldOutTrainingData?: boolean;
|
|
36
|
+
/** Require reward to be strictly greater than this value. Default 0. */
|
|
37
|
+
minimumQualityExclusive?: number;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* What a signed-signal exporter (verifiers, RFT) does with lines that are not
|
|
41
|
+
* clean trainable successes — realness-gated lines above all.
|
|
42
|
+
*
|
|
43
|
+
* - 'exclude' — the default, the same fail-closed policy as every
|
|
44
|
+
* other training export: positive, completed,
|
|
45
|
+
* non-gated rows on a trainable split.
|
|
46
|
+
* - 'zero-and-flag' — keep them, at their non-positive (or null) reward,
|
|
47
|
+
* with `RealnessLabels` on the row. The dataset release
|
|
48
|
+
* sets this per `FORMAT_GATE_DISPOSITION`: in these
|
|
49
|
+
* formats the reward is a signed learning signal, so a
|
|
50
|
+
* gamed trajectory at reward 0 is a correct negative,
|
|
51
|
+
* and dropping it would bias the negative population
|
|
52
|
+
* toward honest failures and leave a trainer no example
|
|
53
|
+
* of gaming being penalized. The split policy is NOT
|
|
54
|
+
* relaxed: held-out lines still need the named opt-in.
|
|
55
|
+
*
|
|
56
|
+
* SFT deliberately has no such option — an SFT row is an imitation target and
|
|
57
|
+
* a gamed trajectory must never appear in one at any weight.
|
|
58
|
+
*/
|
|
59
|
+
type GatedLineDisposition = 'exclude' | 'zero-and-flag';
|
|
60
|
+
interface SignedSignalExportOptions extends TrainingExportOptions {
|
|
61
|
+
/** Disposition for non-trainable lines. Default 'exclude'. */
|
|
62
|
+
gatedLines?: GatedLineDisposition;
|
|
63
|
+
}
|
|
64
|
+
type SftExportOptions = TrainingExportOptions;
|
|
65
|
+
interface SftRow {
|
|
66
|
+
messages: ChatMessage[];
|
|
67
|
+
metadata: {
|
|
68
|
+
rollout_id: string;
|
|
69
|
+
run_id: string;
|
|
70
|
+
candidate_id: string | null;
|
|
71
|
+
instance_id: string;
|
|
72
|
+
reward: number;
|
|
73
|
+
} & RealnessLabels;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Supervised fine-tune rows: the completed conversation of each qualifying
|
|
77
|
+
* line. Fail-closed filters: trainable split only (never holdout/canary),
|
|
78
|
+
* reward strictly above `minimumQualityExclusive` (default 0), realness-gated
|
|
79
|
+
* lines never qualify, gap lines carry no trainable content, and
|
|
80
|
+
* copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001
|
|
81
|
+
* rule 7 — see `ChatMessage.is_copied_context`).
|
|
82
|
+
*
|
|
83
|
+
* `realness_gated` is therefore always `false` on an emitted row. It is carried
|
|
84
|
+
* anyway: an SFT row is a pure imitation target, so the row states its realness
|
|
85
|
+
* claims instead of making the reader know the format's policy, and carrying
|
|
86
|
+
* both flags on all four shapes is what lets the release accounting measure
|
|
87
|
+
* every config with one rule rather than skipping the one whose row shape
|
|
88
|
+
* happened to omit the field.
|
|
89
|
+
*/
|
|
90
|
+
declare function toSftRows(lines: MintedRolloutLine[], options?: SftExportOptions): SftRow[];
|
|
91
|
+
interface RewardRow {
|
|
92
|
+
/** First user turn — the task prompt. */
|
|
93
|
+
prompt: string;
|
|
94
|
+
steps: RolloutStep[];
|
|
95
|
+
reward: number;
|
|
96
|
+
metadata: {
|
|
97
|
+
rollout_id: string;
|
|
98
|
+
run_id: string;
|
|
99
|
+
candidate_id: string | null;
|
|
100
|
+
instance_id: string;
|
|
101
|
+
split: RolloutSplit;
|
|
102
|
+
} & RealnessLabels;
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* Reward-labeled rows for completed, positive-quality training runs.
|
|
106
|
+
*/
|
|
107
|
+
declare function toRewardRows(lines: MintedRolloutLine[], options?: TrainingExportOptions): RewardRow[];
|
|
108
|
+
interface VerifiersTokenUsage {
|
|
109
|
+
input_tokens: number | null;
|
|
110
|
+
output_tokens: number | null;
|
|
111
|
+
reasoning_tokens: number | null;
|
|
112
|
+
cache_read_tokens: number | null;
|
|
113
|
+
cache_write_tokens: number | null;
|
|
114
|
+
}
|
|
115
|
+
interface VerifiersRolloutOutput {
|
|
116
|
+
/** Messages through the last turn BEFORE the first assistant turn. */
|
|
117
|
+
prompt: ChatMessage[];
|
|
118
|
+
/** The first assistant turn onward — what the policy produced. */
|
|
119
|
+
completion: ChatMessage[];
|
|
120
|
+
reward: number | null;
|
|
121
|
+
metrics: Record<string, unknown>;
|
|
122
|
+
tool_defs: ToolDef[];
|
|
123
|
+
token_usage: VerifiersTokenUsage;
|
|
124
|
+
info: {
|
|
125
|
+
task: RolloutLine['task'];
|
|
126
|
+
policy: RolloutLine['policy'];
|
|
127
|
+
rollout_id: string;
|
|
128
|
+
run_id: string;
|
|
129
|
+
experiment_id: string | null;
|
|
130
|
+
candidate_id: string | null;
|
|
131
|
+
generation: number | null;
|
|
132
|
+
candidate_index: number | null;
|
|
133
|
+
role: RolloutLine['role'];
|
|
134
|
+
} & RealnessLabels;
|
|
135
|
+
}
|
|
136
|
+
declare function toVerifiersRolloutOutput(line: MintedRolloutLine): VerifiersRolloutOutput;
|
|
137
|
+
declare function toVerifiersRolloutOutputs(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): VerifiersRolloutOutput[];
|
|
138
|
+
interface RftItem {
|
|
139
|
+
/** Prompt turns only — the graded completion is re-sampled during RFT. */
|
|
140
|
+
messages: ChatMessage[];
|
|
141
|
+
/** Verdict/label fields the grader references as item.reference.* */
|
|
142
|
+
reference: {
|
|
143
|
+
reward: number | null;
|
|
144
|
+
reward_source: string | null;
|
|
145
|
+
verdict: unknown;
|
|
146
|
+
instance_id: string;
|
|
147
|
+
suite: string;
|
|
148
|
+
split: RolloutSplit;
|
|
149
|
+
rollout_id: string;
|
|
150
|
+
} & RealnessLabels;
|
|
151
|
+
}
|
|
152
|
+
declare function toRftItem(line: MintedRolloutLine): RftItem;
|
|
153
|
+
/** RFT needs a real prompt: lines whose transcript starts with prompt turns. */
|
|
154
|
+
declare function toRftItems(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): RftItem[];
|
|
155
|
+
declare function toJsonl(rows: ReadonlyArray<unknown>): string;
|
|
156
|
+
//#endregion
|
|
157
|
+
//#region src/rollout/gate-checks.d.ts
|
|
158
|
+
/**
|
|
159
|
+
* Every gate check in the package, in the order they are applied.
|
|
160
|
+
*
|
|
161
|
+
* Order is load-bearing only for which message a caller sees first: a line that
|
|
162
|
+
* trips two checks reports the earlier one, and `reward-relationship` is first
|
|
163
|
+
* because it is the invariant the other three protect.
|
|
164
|
+
*/
|
|
165
|
+
declare const GATE_CHECK_IDS: readonly ['reward-relationship', 'gated-evidence', 'undeclared-step-payload', 'unscreened-reward'];
|
|
166
|
+
type GateCheckId = (typeof GATE_CHECK_IDS)[number];
|
|
167
|
+
/**
|
|
168
|
+
* An outcome as it reaches a check.
|
|
169
|
+
*
|
|
170
|
+
* Deliberately accepts a raw record as well as the typed shape: the checks are
|
|
171
|
+
* the RUNTIME half of the gate, and the callers they exist for — JSON off a
|
|
172
|
+
* ledger, a plain-JavaScript consumer of the published package — arrive with no
|
|
173
|
+
* types at all. `Partial` because a tripwire states only the fields it trips on.
|
|
174
|
+
*/
|
|
175
|
+
type GateCheckedOutcome = Partial<RolloutOutcome> | Readonly<Record<string, unknown>>;
|
|
176
|
+
/**
|
|
177
|
+
* What a gate check reads: the reward-bearing surface of ONE LINE.
|
|
178
|
+
*
|
|
179
|
+
* For three rounds the subject was the OUTCOME alone, and that assumption is
|
|
180
|
+
* what produced the next leak rather than any missing check: `steps[]` sits on
|
|
181
|
+
* the LINE, outside `outcome`, so a per-step reward on a gated line was read by
|
|
182
|
+
* no check at all while `toRewardRows` copied it out verbatim — through the
|
|
183
|
+
* MINTED door, not merely the raw one. Widening the subject is what makes
|
|
184
|
+
* "somewhere else on the line" a place the checks can see.
|
|
185
|
+
*
|
|
186
|
+
* `outcome` is REQUIRED, and that is the point: a bare `RolloutOutcome` is then
|
|
187
|
+
* not assignable to a subject, so every call site that used to pass one is a
|
|
188
|
+
* COMPILE error until it passes the line instead. A subject with an optional
|
|
189
|
+
* `outcome` would have let the old call sites keep compiling while silently
|
|
190
|
+
* checking nothing — the exact failure this module exists to make impossible.
|
|
191
|
+
*/
|
|
192
|
+
interface GateSubject {
|
|
193
|
+
outcome: GateCheckedOutcome;
|
|
194
|
+
/** The line's trajectory steps, when it carries any. */
|
|
195
|
+
steps?: unknown;
|
|
196
|
+
}
|
|
197
|
+
interface GateCheck {
|
|
198
|
+
id: GateCheckId;
|
|
199
|
+
/** One sentence: what this check refuses. */
|
|
200
|
+
refuses: string;
|
|
201
|
+
/** One dotted-path message per defect; `[]` when the line is clean. */
|
|
202
|
+
errors: (subject: GateSubject) => string[];
|
|
203
|
+
/**
|
|
204
|
+
* Every minimal subject that MUST trip `errors` — the executable form of
|
|
205
|
+
* `refuses`, and the reason a check cannot be added without being provable.
|
|
206
|
+
* The calibration test feeds each one to every entry point declaring
|
|
207
|
+
* `enforced`.
|
|
208
|
+
*
|
|
209
|
+
* A LIST rather than one case: a check that refuses two distinct populations
|
|
210
|
+
* (a positive reward AND a reward it cannot read as a number) proved able to
|
|
211
|
+
* hold for the first while silently passing the second, so each population
|
|
212
|
+
* states its own tripwire and each is exercised separately.
|
|
213
|
+
*/
|
|
214
|
+
tripwires: GateSubject[];
|
|
215
|
+
}
|
|
216
|
+
/**
|
|
217
|
+
* The reward-bearing outcome fields that are NOT the scalar: the numbers the
|
|
218
|
+
* reward was computed from, and the verdict record that claimed it.
|
|
219
|
+
*
|
|
220
|
+
* Returned as one block rather than filtered key-by-key. A key-name heuristic
|
|
221
|
+
* ("zero anything matching `layer.*` or `/score/`") is the same defect shape as
|
|
222
|
+
* the line-oriented regex the AST score guard replaced: it holds until someone
|
|
223
|
+
* names a metric `pass_fraction`, and the next reward-shaped key ships at full
|
|
224
|
+
* value. The producer's OWN classification — "this is the scalar, that is
|
|
225
|
+
* everything else" — is the only partition that cannot be out-guessed.
|
|
226
|
+
*/
|
|
227
|
+
declare function gatedEvidenceOf(subject: GateSubject): GatedEvidence | undefined;
|
|
228
|
+
/**
|
|
229
|
+
* The registry. Total over `GateCheckId`, so an id with no check does not
|
|
230
|
+
* compile, and `GATE_CHECK_IDS` stays the single enumeration everything
|
|
231
|
+
* iterates.
|
|
232
|
+
*/
|
|
233
|
+
declare const GATE_CHECKS: { readonly [K in GateCheckId]: GateCheck; };
|
|
234
|
+
/**
|
|
235
|
+
* What ONE entry point does about ONE check.
|
|
236
|
+
*
|
|
237
|
+
* `repair` and `omit` both carry a mandatory sentence, which is the mechanism
|
|
238
|
+
* that keeps a legitimate omission distinguishable from a forgotten one: you
|
|
239
|
+
* cannot skip a check without writing down why, and the reasons are readable
|
|
240
|
+
* side by side in `GATE_POLICIES`.
|
|
241
|
+
*/
|
|
242
|
+
type GateCheckDisposition = {
|
|
243
|
+
readonly kind: 'enforce';
|
|
244
|
+
} |
|
|
245
|
+
/** Resolved by TRANSFORMING the line instead of rejecting it; `by` names the function. */
|
|
246
|
+
{
|
|
247
|
+
readonly kind: 'repair';
|
|
248
|
+
readonly by: string;
|
|
249
|
+
} |
|
|
250
|
+
/** Deliberately not applied here; `because` states the reason. */
|
|
251
|
+
{
|
|
252
|
+
readonly kind: 'omit';
|
|
253
|
+
readonly because: string;
|
|
254
|
+
};
|
|
255
|
+
/** Total over `GateCheckId`: a new check makes every policy literal a type error. */
|
|
256
|
+
type GatePolicy = { readonly [K in GateCheckId]: GateCheckDisposition; };
|
|
257
|
+
/**
|
|
258
|
+
* Every entry point that decides about the gate, and what it decides.
|
|
259
|
+
*
|
|
260
|
+
* Read this as the package's gate policy in one screen. The four entry points
|
|
261
|
+
* are not interchangeable — a validator that rejects, a mint funnel that
|
|
262
|
+
* repairs, a runtime backstop for untyped callers, and a release certifier over
|
|
263
|
+
* emitted rows — and the dispositions say which is which.
|
|
264
|
+
*/
|
|
265
|
+
declare const GATE_POLICIES: {
|
|
266
|
+
/**
|
|
267
|
+
* The schema validator. Rejects the reward relationship and NOTHING ELSE, on
|
|
268
|
+
* purpose: it runs on every line read off disk, and the other two conditions
|
|
269
|
+
* describe artifacts that already exist.
|
|
270
|
+
*/
|
|
271
|
+
readonly validateRolloutLine: {
|
|
272
|
+
readonly 'reward-relationship': {
|
|
273
|
+
readonly kind: 'enforce';
|
|
274
|
+
};
|
|
275
|
+
readonly 'gated-evidence': GateCheckDisposition;
|
|
276
|
+
readonly 'undeclared-step-payload': GateCheckDisposition;
|
|
277
|
+
readonly 'unscreened-reward': GateCheckDisposition;
|
|
278
|
+
};
|
|
279
|
+
/**
|
|
280
|
+
* The mint funnel — the single door every `MintedRolloutLine` passes. Validates
|
|
281
|
+
* first (so the reward relationship has already been rejected), then refuses
|
|
282
|
+
* what cannot be repaired, then repairs what can.
|
|
283
|
+
*/
|
|
284
|
+
readonly assertMinted: {
|
|
285
|
+
readonly 'reward-relationship': {
|
|
286
|
+
readonly kind: 'enforce';
|
|
287
|
+
};
|
|
288
|
+
readonly 'gated-evidence': GateCheckDisposition;
|
|
289
|
+
readonly 'undeclared-step-payload': GateCheckDisposition;
|
|
290
|
+
readonly 'unscreened-reward': {
|
|
291
|
+
readonly kind: 'enforce';
|
|
292
|
+
};
|
|
293
|
+
};
|
|
294
|
+
/**
|
|
295
|
+
* The runtime backstop, and the one entry point with no license to omit
|
|
296
|
+
* anything: it exists for callers the type system never saw (plain JavaScript
|
|
297
|
+
* handing an object literal to a published exporter), so a check it skips is a
|
|
298
|
+
* check that does not run at all for them. This is where the fourth leak was.
|
|
299
|
+
*/
|
|
300
|
+
readonly assertRewardGate: {
|
|
301
|
+
readonly 'reward-relationship': {
|
|
302
|
+
readonly kind: 'enforce';
|
|
303
|
+
};
|
|
304
|
+
readonly 'gated-evidence': {
|
|
305
|
+
readonly kind: 'enforce';
|
|
306
|
+
};
|
|
307
|
+
readonly 'undeclared-step-payload': {
|
|
308
|
+
readonly kind: 'enforce';
|
|
309
|
+
};
|
|
310
|
+
readonly 'unscreened-reward': {
|
|
311
|
+
readonly kind: 'enforce';
|
|
312
|
+
};
|
|
313
|
+
};
|
|
314
|
+
/**
|
|
315
|
+
* The release certifier. Same checks, measured over the rows a release is
|
|
316
|
+
* ABOUT TO WRITE rather than over one line's outcome — see `REPORT_MEASURES`
|
|
317
|
+
* in `release/gate-report.ts`, which is the second total map this policy
|
|
318
|
+
* drives.
|
|
319
|
+
*/
|
|
320
|
+
readonly assertGateReport: {
|
|
321
|
+
readonly 'reward-relationship': {
|
|
322
|
+
readonly kind: 'enforce';
|
|
323
|
+
};
|
|
324
|
+
readonly 'gated-evidence': {
|
|
325
|
+
readonly kind: 'enforce';
|
|
326
|
+
};
|
|
327
|
+
readonly 'undeclared-step-payload': {
|
|
328
|
+
readonly kind: 'enforce';
|
|
329
|
+
};
|
|
330
|
+
readonly 'unscreened-reward': {
|
|
331
|
+
readonly kind: 'enforce';
|
|
332
|
+
};
|
|
333
|
+
};
|
|
334
|
+
};
|
|
335
|
+
/** Every entry point that declares a gate policy. */
|
|
336
|
+
type GateEntryPoint = keyof typeof GATE_POLICIES;
|
|
337
|
+
/**
|
|
338
|
+
* Run the checks one entry point enforces. The ONLY way an entry point should
|
|
339
|
+
* obtain gate errors — hand-composing two of the three is the bug this module
|
|
340
|
+
* exists to remove.
|
|
341
|
+
*/
|
|
342
|
+
declare function gateErrors(subject: GateSubject, policy: GatePolicy): string[];
|
|
343
|
+
//#endregion
|
|
344
|
+
//#region src/rollout/interchange/harbor.d.ts
|
|
345
|
+
declare const ATIF_SCHEMA_VERSION = "ATIF-v1.7";
|
|
346
|
+
/** Gap note on every imported line — ATIF carries no verdict, so nothing is scored. */
|
|
347
|
+
declare const HARBOR_IMPORT_GAP = "imported from Harbor ATIF; no verdict";
|
|
348
|
+
type HarborStepSource = 'system' | 'user' | 'agent';
|
|
349
|
+
interface HarborImageSource {
|
|
350
|
+
media_type: string;
|
|
351
|
+
path: string;
|
|
352
|
+
}
|
|
353
|
+
interface HarborContentPart {
|
|
354
|
+
type: 'text' | 'image';
|
|
355
|
+
text?: string;
|
|
356
|
+
source?: HarborImageSource;
|
|
357
|
+
}
|
|
358
|
+
interface HarborToolCall {
|
|
359
|
+
tool_call_id: string;
|
|
360
|
+
function_name: string;
|
|
361
|
+
/** ATIF requires a decoded JSON object here, unlike our raw argument string. */
|
|
362
|
+
arguments: Record<string, unknown>;
|
|
363
|
+
extra?: Record<string, unknown>;
|
|
364
|
+
}
|
|
365
|
+
interface HarborSubagentTrajectoryRef {
|
|
366
|
+
trajectory_id?: string;
|
|
367
|
+
trajectory_path?: string;
|
|
368
|
+
/** Informational only since v1.7 — never a resolution key. */
|
|
369
|
+
session_id?: string;
|
|
370
|
+
extra?: Record<string, unknown>;
|
|
371
|
+
}
|
|
372
|
+
interface HarborObservationResult {
|
|
373
|
+
source_call_id?: string;
|
|
374
|
+
content?: string | HarborContentPart[];
|
|
375
|
+
subagent_trajectory_ref?: HarborSubagentTrajectoryRef[];
|
|
376
|
+
extra?: Record<string, unknown>;
|
|
377
|
+
}
|
|
378
|
+
interface HarborObservation {
|
|
379
|
+
results: HarborObservationResult[];
|
|
380
|
+
}
|
|
381
|
+
interface HarborMetrics {
|
|
382
|
+
prompt_tokens?: number;
|
|
383
|
+
completion_tokens?: number;
|
|
384
|
+
cached_tokens?: number;
|
|
385
|
+
cost_usd?: number;
|
|
386
|
+
prompt_token_ids?: number[];
|
|
387
|
+
completion_token_ids?: number[];
|
|
388
|
+
logprobs?: number[];
|
|
389
|
+
extra?: Record<string, unknown>;
|
|
390
|
+
}
|
|
391
|
+
interface HarborStep {
|
|
392
|
+
/** Ordinal, sequential from 1. */
|
|
393
|
+
step_id: number;
|
|
394
|
+
timestamp?: string;
|
|
395
|
+
source: HarborStepSource;
|
|
396
|
+
model_name?: string;
|
|
397
|
+
reasoning_effort?: string | number;
|
|
398
|
+
message: string | HarborContentPart[];
|
|
399
|
+
reasoning_content?: string;
|
|
400
|
+
tool_calls?: HarborToolCall[];
|
|
401
|
+
observation?: HarborObservation;
|
|
402
|
+
metrics?: HarborMetrics;
|
|
403
|
+
llm_call_count?: number;
|
|
404
|
+
is_copied_context?: boolean;
|
|
405
|
+
extra?: Record<string, unknown>;
|
|
406
|
+
}
|
|
407
|
+
interface HarborAgent {
|
|
408
|
+
name: string;
|
|
409
|
+
version: string;
|
|
410
|
+
model_name?: string;
|
|
411
|
+
/** OpenAI function-calling schema — byte-identical to our `ToolDef`. */
|
|
412
|
+
tool_definitions?: ToolDef[];
|
|
413
|
+
extra?: Record<string, unknown>;
|
|
414
|
+
}
|
|
415
|
+
interface HarborFinalMetrics {
|
|
416
|
+
total_prompt_tokens?: number;
|
|
417
|
+
total_completion_tokens?: number;
|
|
418
|
+
total_cached_tokens?: number;
|
|
419
|
+
total_cost_usd?: number;
|
|
420
|
+
total_steps?: number;
|
|
421
|
+
extra?: Record<string, unknown>;
|
|
422
|
+
}
|
|
423
|
+
interface HarborTrajectory {
|
|
424
|
+
schema_version: string;
|
|
425
|
+
session_id?: string;
|
|
426
|
+
/** Required on embedded subagents; we always set it so lines stay joinable. */
|
|
427
|
+
trajectory_id?: string;
|
|
428
|
+
agent: HarborAgent;
|
|
429
|
+
steps: HarborStep[];
|
|
430
|
+
notes?: string;
|
|
431
|
+
final_metrics?: HarborFinalMetrics;
|
|
432
|
+
continued_trajectory_ref?: string;
|
|
433
|
+
subagent_trajectories?: HarborTrajectory[];
|
|
434
|
+
extra?: Record<string, unknown>;
|
|
435
|
+
}
|
|
436
|
+
/**
|
|
437
|
+
* Assemble one episode's flat lines into a single ATIF trajectory tree,
|
|
438
|
+
* linked by `parent_rollout_id`.
|
|
439
|
+
*
|
|
440
|
+
* Reward, verdict and split are NOT emitted (ATIF models none of them); the
|
|
441
|
+
* split and the rest of the task coordinates survive only in `extra.tangle`.
|
|
442
|
+
*
|
|
443
|
+
* We deliberately do NOT synthesize an `observation.subagent_trajectory_ref`
|
|
444
|
+
* pointing at each child: our ledger records WHICH invocation spawned a
|
|
445
|
+
* worker, not which STEP did, and attaching the ref to a guessed step would
|
|
446
|
+
* fabricate a causal claim. Children are embedded in `subagent_trajectories`
|
|
447
|
+
* (each with the `trajectory_id` the spec requires) and the edge is stated in
|
|
448
|
+
* the child's escrowed `parent_rollout_id`.
|
|
449
|
+
*
|
|
450
|
+
* Throws when the lines are not one tree — use `toHarborTrajectories` for a forest.
|
|
451
|
+
*/
|
|
452
|
+
declare function toHarborTrajectory(lines: RolloutLine[]): HarborTrajectory;
|
|
453
|
+
/** Every independent tree in the input, one ATIF document each. */
|
|
454
|
+
declare function toHarborTrajectories(lines: RolloutLine[]): HarborTrajectory[];
|
|
455
|
+
interface FromHarborOptions {
|
|
456
|
+
/** Injected clock for deterministic output when the source carries no capture time. */
|
|
457
|
+
now?: () => Date;
|
|
458
|
+
}
|
|
459
|
+
/**
|
|
460
|
+
* Flatten an ATIF trajectory tree back into `tangle.rollout.v1` lines, parent
|
|
461
|
+
* first, each child carrying `parent_rollout_id`.
|
|
462
|
+
*
|
|
463
|
+
* Every line comes back UNLABELED: `reward`, `reward_source` and `verdict` are
|
|
464
|
+
* null and `provenance.gap` says why. ATIF models no verdict, so scoring an
|
|
465
|
+
* imported trajectory is a judge's job, not this function's. Every line lands
|
|
466
|
+
* on `holdout` whatever the document claims — see `relabelImportedSplit`.
|
|
467
|
+
*/
|
|
468
|
+
declare function fromHarborTrajectory(trajectory: HarborTrajectory, options?: FromHarborOptions): RolloutLine[];
|
|
469
|
+
/**
|
|
470
|
+
* THE explicit door out of `holdout` for imported lines.
|
|
471
|
+
*
|
|
472
|
+
* Import forces `holdout` because a document's own claim about its split is not
|
|
473
|
+
* evidence — anyone can write `extra.tangle.task.split`. Promoting a file to a
|
|
474
|
+
* trainable split is an operator's decision about provenance they verified, so
|
|
475
|
+
* it is a separate, greppable call: `grep relabelImportedSplit` enumerates
|
|
476
|
+
* every place foreign data was declared trainable, which is exactly the audit
|
|
477
|
+
* the trusted-escrow version made impossible.
|
|
478
|
+
*
|
|
479
|
+
* Returns plain `RolloutLine`s. They still have to pass `assertMinted` (and its
|
|
480
|
+
* anti-Goodhart check) to reach an exporter — re-labeling a split is not
|
|
481
|
+
* minting a reward.
|
|
482
|
+
*/
|
|
483
|
+
declare function relabelImportedSplit(lines: readonly RolloutLine[], split: RolloutSplit): RolloutLine[];
|
|
484
|
+
//#endregion
|
|
485
|
+
//#region src/rollout/ledger.d.ts
|
|
486
|
+
/** Replace the ledger file with exactly `lines`. */
|
|
487
|
+
declare function writeRolloutLedger(path: string, lines: RolloutLine[]): Promise<void>;
|
|
488
|
+
/** Append `lines` to the ledger file (created if absent). */
|
|
489
|
+
declare function appendRolloutLines(path: string, lines: RolloutLine[]): Promise<void>;
|
|
490
|
+
/**
|
|
491
|
+
* Read and validate every line. Throws on the first malformed/invalid line
|
|
492
|
+
* (with its 1-based line number) — fail-closed, never a silent drop.
|
|
493
|
+
*
|
|
494
|
+
* Validation includes the anti-Goodhart invariant, which is why the result is
|
|
495
|
+
* `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this
|
|
496
|
+
* process from outside the type system (another run, another machine, a
|
|
497
|
+
* hand-edited JSONL), so this read is the runtime boundary where a poisoned
|
|
498
|
+
* line is refused rather than exported.
|
|
499
|
+
*/
|
|
500
|
+
declare function readRolloutLedger(path: string): Promise<MintedRolloutLine[]>;
|
|
501
|
+
/**
|
|
502
|
+
* Read a ledger under the WRITE-side policy (`validateRolloutLine`), which
|
|
503
|
+
* omits the unscreened-reward check. `writeRolloutLedger` accepts a
|
|
504
|
+
* supervision-journal row (`realness_screened: false` with a positive reward
|
|
505
|
+
* — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says
|
|
506
|
+
* such rows "must stay writable, readable and reportable"; a read API that
|
|
507
|
+
* only re-validated under `assertMinted` made every such file unreadable —
|
|
508
|
+
* write-accepted but read-refused is a data-loss trap.
|
|
509
|
+
*
|
|
510
|
+
* The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here
|
|
511
|
+
* can reach a training exporter without passing `assertMinted`, so the
|
|
512
|
+
* promotion gate (which DOES enforce unscreened-reward) is exactly as closed
|
|
513
|
+
* as before. Use `readRolloutLedger` when the file is training data.
|
|
514
|
+
*/
|
|
515
|
+
declare function readRolloutJournal(path: string): Promise<RolloutLine[]>;
|
|
516
|
+
//#endregion
|
|
517
|
+
//#region src/rollout/reward.d.ts
|
|
518
|
+
/**
|
|
519
|
+
* Which split's score wins when a record carries both. `'holdout'` is the
|
|
520
|
+
* canonical "real signal" default; `'search'` exists because some callers
|
|
521
|
+
* deliberately score on the search split when both are present.
|
|
522
|
+
*/
|
|
523
|
+
type ScorePreference = 'holdout' | 'search';
|
|
524
|
+
/** Only the outcome is read, so every accessor here accepts anything carrying one. */
|
|
525
|
+
type Scored = Pick<RunRecord, 'outcome'>;
|
|
526
|
+
/** True when the authenticity gate flagged the run as gamed (`realness.gated`). */
|
|
527
|
+
declare function isRealnessGated(record: Scored): boolean;
|
|
528
|
+
/**
|
|
529
|
+
* The RAW score recorded on ONE split, with no cross-split fallback and no
|
|
530
|
+
* anti-Goodhart gate.
|
|
531
|
+
*
|
|
532
|
+
* The narrowest of the three raw readers, and the one every split-scoped
|
|
533
|
+
* consumer wants: a per-split report, a promotion gate, or a paired comparison
|
|
534
|
+
* asks "what did this run score on the split I am summarising", and answering
|
|
535
|
+
* it with the other split's number silently mixes populations. `undefined` =
|
|
536
|
+
* that split was never scored.
|
|
537
|
+
*
|
|
538
|
+
* Same warning as `observedScore`: this INCLUDES runs flagged as gamed. Never
|
|
539
|
+
* feed it into training data.
|
|
540
|
+
*/
|
|
541
|
+
declare function observedSplitScore(record: Scored, split: ScorePreference): number | undefined;
|
|
542
|
+
/**
|
|
543
|
+
* The RAW split score the run carries, with NO anti-Goodhart gate applied.
|
|
544
|
+
*
|
|
545
|
+
* INCLUDES RUNS FLAGGED AS GAMED (`outcome.realness.gated === true`); NEVER
|
|
546
|
+
* feed this into training data — a fine-tune that sees it learns from gamed
|
|
547
|
+
* successes. It is exported anyway because analysis, reporting, and
|
|
548
|
+
* reward-hacking detection legitimately need the ungated number: forcing a
|
|
549
|
+
* gamed run to 0 collapses the proxy signal toward ground truth and makes a
|
|
550
|
+
* detector report "clean" on exactly the population that is being gamed.
|
|
551
|
+
*
|
|
552
|
+
* Returns `undefined` when the record carries neither score — an unscored run
|
|
553
|
+
* is a labeled gap, not a measured zero, and each caller picks its own
|
|
554
|
+
* sentinel (`?? 0`, `?? null`, skip, throw). Non-finite values are returned
|
|
555
|
+
* as-is; callers that care keep their own `Number.isFinite` guard.
|
|
556
|
+
*/
|
|
557
|
+
declare function observedScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
558
|
+
/** Which split actually carried the score, or that none did. */
|
|
559
|
+
type ScoreOrigin = 'holdout' | 'search' | 'unscored';
|
|
560
|
+
/**
|
|
561
|
+
* Where `observedScore` / `trainingScore` read their number from — the
|
|
562
|
+
* provenance label a rollout line's `reward_source` is built from, and the
|
|
563
|
+
* only supported way to ask "was this run scored at all" without respelling
|
|
564
|
+
* the field access.
|
|
565
|
+
*/
|
|
566
|
+
declare function scoreOrigin(record: Scored, prefer?: ScorePreference): ScoreOrigin;
|
|
567
|
+
/**
|
|
568
|
+
* The GATED score — the only derivation allowed to reach training data.
|
|
569
|
+
*
|
|
570
|
+
* A realness-gated run scores 0 no matter what it claims, so a fine-tune
|
|
571
|
+
* cannot learn from a gamed success. An unscored run stays `undefined` (a
|
|
572
|
+
* labeled gap), keeping "we never measured this" distinct from "we measured
|
|
573
|
+
* zero"; callers that need a number apply their own sentinel.
|
|
574
|
+
*/
|
|
575
|
+
declare function trainingScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
576
|
+
/**
|
|
577
|
+
* `{reward, gated}` as written onto a minted `RolloutLine` — `trainingScore`
|
|
578
|
+
* plus the flag itself, so the gate travels into the exported row and a
|
|
579
|
+
* downstream filter can drop or down-weight the line.
|
|
580
|
+
*
|
|
581
|
+
* An unscored record yields `reward: null`, matching the schema's "no verdict
|
|
582
|
+
* exists — a labeled gap, never 0" rule. It previously collapsed to 0, which
|
|
583
|
+
* made a run nobody graded indistinguishable from one graded as a total
|
|
584
|
+
* failure, and taught any trainer reading the row that the trajectory was bad.
|
|
585
|
+
* A gated run still yields 0, because that IS a verdict: the gate decided.
|
|
586
|
+
*/
|
|
587
|
+
declare function trainingReward(record: Scored): {
|
|
588
|
+
reward: number | null;
|
|
589
|
+
gated: boolean;
|
|
590
|
+
};
|
|
591
|
+
//#endregion
|
|
592
|
+
//#region src/rollout/mint.d.ts
|
|
593
|
+
/** Redactor applied to every exported string (secrets, PII). Identity by default. */
|
|
594
|
+
type RolloutScrubber = (text: string) => string;
|
|
595
|
+
interface MintRolloutOptions {
|
|
596
|
+
scrub?: RolloutScrubber;
|
|
597
|
+
/** Cap steps per line (longest runs first drop middle steps). Default: no cap. */
|
|
598
|
+
maxSteps?: number;
|
|
599
|
+
/** Role recorded on every minted line. Default 'agent' (a solo eval run). */
|
|
600
|
+
role?: RolloutRole;
|
|
601
|
+
/** Task suite label. Default: the record's `experimentId`. */
|
|
602
|
+
suite?: string;
|
|
603
|
+
/** Injected clock for deterministic output. */
|
|
604
|
+
now?: () => Date;
|
|
605
|
+
}
|
|
606
|
+
interface MintRolloutResult {
|
|
607
|
+
rows: MintedRolloutLine[];
|
|
608
|
+
/** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
|
|
609
|
+
missingTraces: string[];
|
|
610
|
+
}
|
|
611
|
+
/**
|
|
612
|
+
* Join RunRecords with their traces into canonical rollout lines. Records
|
|
613
|
+
* without spans are emitted as labeled gap lines and reported in
|
|
614
|
+
* `missingTraces`. Execution-only records without a task score are rejected
|
|
615
|
+
* because a missing training label is not a zero reward.
|
|
616
|
+
*/
|
|
617
|
+
declare function mintRolloutRows(records: RunRecord[], store: TraceStore, options?: MintRolloutOptions): Promise<MintRolloutResult>;
|
|
618
|
+
//#endregion
|
|
619
|
+
//#region src/rollout/readers/claude-jsonl.d.ts
|
|
620
|
+
declare const DEFAULT_CLAUDE_PROJECTS_DIR: string;
|
|
621
|
+
/** Claude Code's project-directory slug for a working directory. */
|
|
622
|
+
declare function claudeProjectSlug(cwd: string): string;
|
|
623
|
+
interface ClaudeTranscriptRef {
|
|
624
|
+
sessionId: string;
|
|
625
|
+
path: string;
|
|
626
|
+
}
|
|
627
|
+
/** Transcript files recorded for sessions launched from `cwd`. */
|
|
628
|
+
declare function findClaudeTranscripts(cwd: string, projectsDir?: string): Promise<ClaudeTranscriptRef[]>;
|
|
629
|
+
interface ClaudeUsageTotals {
|
|
630
|
+
tokensIn: number;
|
|
631
|
+
tokensOut: number;
|
|
632
|
+
cacheRead: number;
|
|
633
|
+
cacheWrite: number;
|
|
634
|
+
}
|
|
635
|
+
interface ClaudeTranscript {
|
|
636
|
+
messages: ChatMessage[];
|
|
637
|
+
usage: ClaudeUsageTotals;
|
|
638
|
+
/** Timestamp of the first conversation line; null = empty transcript. */
|
|
639
|
+
startedAt: string | null;
|
|
640
|
+
endedAt: string | null;
|
|
641
|
+
model: string | null;
|
|
642
|
+
}
|
|
643
|
+
interface ReadClaudeTranscriptOptions {
|
|
644
|
+
/**
|
|
645
|
+
* Read the sidechain (subagent) thread instead of skipping it. Subagent
|
|
646
|
+
* transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain
|
|
647
|
+
* lines end to end, so their usage is invisible without this.
|
|
648
|
+
*/
|
|
649
|
+
readonly includeSidechain?: boolean;
|
|
650
|
+
}
|
|
651
|
+
/** Parse one transcript jsonl into canonical messages + usage totals. */
|
|
652
|
+
declare function readClaudeTranscript(path: string, options?: ReadClaudeTranscriptOptions): Promise<ClaudeTranscript>;
|
|
653
|
+
//#endregion
|
|
654
|
+
//#region src/rollout/readers/opencode-sqlite.d.ts
|
|
655
|
+
declare const DEFAULT_OPENCODE_DB: string;
|
|
656
|
+
interface OpencodeSessionRow {
|
|
657
|
+
id: string;
|
|
658
|
+
parentId: string | null;
|
|
659
|
+
directory: string;
|
|
660
|
+
agent: string | null;
|
|
661
|
+
/** Raw session.model JSON: {id, providerID, variant} where present. */
|
|
662
|
+
model: {
|
|
663
|
+
id?: string;
|
|
664
|
+
providerID?: string;
|
|
665
|
+
} | null;
|
|
666
|
+
costUsd: number;
|
|
667
|
+
tokensInput: number;
|
|
668
|
+
tokensOutput: number;
|
|
669
|
+
tokensReasoning: number;
|
|
670
|
+
tokensCacheRead: number;
|
|
671
|
+
tokensCacheWrite: number;
|
|
672
|
+
timeCreated: number;
|
|
673
|
+
timeUpdated: number;
|
|
674
|
+
}
|
|
675
|
+
/** Open the store read-only; null = unavailable/corrupt (caller records a gap). */
|
|
676
|
+
declare function openOpencodeDb(path?: string): Promise<DatabaseSync | null>;
|
|
677
|
+
/** Sessions whose cwd is `directory` (the worker-clone join key). */
|
|
678
|
+
declare function findOpencodeSessionsByDirectory(db: DatabaseSync, directory: string): OpencodeSessionRow[];
|
|
679
|
+
declare function findOpencodeSessionById(db: DatabaseSync, sessionId: string): OpencodeSessionRow | null;
|
|
680
|
+
/**
|
|
681
|
+
* Convert one session's message+part rows into canonical messages.
|
|
682
|
+
* An opencode assistant message row spans several model steps; each step's
|
|
683
|
+
* parts (reasoning → text → tool …) become one assistant message followed by
|
|
684
|
+
* the role:"tool" results of its calls, preserving order.
|
|
685
|
+
*/
|
|
686
|
+
declare function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[];
|
|
687
|
+
//#endregion
|
|
688
|
+
//#region src/rollout/release/gate-report.d.ts
|
|
689
|
+
/** What a format does with a line the realness gate flagged. */
|
|
690
|
+
type GateDisposition = 'exclude' | 'zero-and-flag';
|
|
691
|
+
declare const FORMAT_GATE_DISPOSITION: Record<ReleaseFormat, GateDisposition>;
|
|
692
|
+
/** What the gate accounting reads off an emitted row, per format. */
|
|
693
|
+
interface ReleaseRowRef {
|
|
694
|
+
rollout_id: string;
|
|
695
|
+
reward: number | null;
|
|
696
|
+
/**
|
|
697
|
+
* The rest of the row that was DERIVED from the reward — the per-layer score
|
|
698
|
+
* dict, the judge verdict record, whatever this format ships beside the
|
|
699
|
+
* scalar. Walked for positive numbers, so the certification is about the
|
|
700
|
+
* whole outcome rather than one field.
|
|
701
|
+
*
|
|
702
|
+
* Absent when the format's row carries nothing but the scalar. NOT the whole
|
|
703
|
+
* row: `cost.tokens_in`, `wall_s` and `total_steps` are positive numbers that
|
|
704
|
+
* have nothing to do with the reward, and a certification that flags them is
|
|
705
|
+
* a certification nobody can act on.
|
|
706
|
+
*/
|
|
707
|
+
evidence?: unknown;
|
|
708
|
+
/**
|
|
709
|
+
* The screen claim AS EMITTED — read off the row, not off the line it came
|
|
710
|
+
* from, because what ships is what matters. Required, not optional: an
|
|
711
|
+
* optional field is how a format quietly opts out of the check that reads it,
|
|
712
|
+
* and every emitted row shape carries `RealnessLabels` precisely so no adapter
|
|
713
|
+
* has to.
|
|
714
|
+
*/
|
|
715
|
+
realness_screened: boolean | null;
|
|
716
|
+
/**
|
|
717
|
+
* The part of an emitted `steps[]` the wire format does not declare.
|
|
718
|
+
*
|
|
719
|
+
* Separate from `evidence` because the declared step fields are FULL of
|
|
720
|
+
* legitimate positive numbers — `durationMs`, `llm_call_count`,
|
|
721
|
+
* `prompt_token_ids` — and a certification that flags those is one nobody can
|
|
722
|
+
* act on. Only the undeclared remainder is unclassified reward-bearing
|
|
723
|
+
* payload, which is the same partition the check applies.
|
|
724
|
+
*
|
|
725
|
+
* Set only by formats whose row carries steps: today `raw` alone.
|
|
726
|
+
*/
|
|
727
|
+
stepEvidence?: unknown;
|
|
728
|
+
}
|
|
729
|
+
/** A positive number found inside an emitted gated row, with where it was. */
|
|
730
|
+
interface EmittedEvidence {
|
|
731
|
+
/** JSON-ish path from the row's evidence root, e.g. `metrics['layer.tests']`. */
|
|
732
|
+
path: string;
|
|
733
|
+
value: number;
|
|
734
|
+
}
|
|
735
|
+
interface FormatGateCounts {
|
|
736
|
+
/** Gated lines that reached this format's exporter. */
|
|
737
|
+
input: number;
|
|
738
|
+
/** Gated rows the format actually wrote. */
|
|
739
|
+
emitted: number;
|
|
740
|
+
/**
|
|
741
|
+
* Gated lines this format did not write. Not all of these are the gate:
|
|
742
|
+
* `verifiers` also drops gap lines (empty transcript) and `rft` drops lines
|
|
743
|
+
* with no prompt turn, so an excluded count can mix both causes.
|
|
744
|
+
*/
|
|
745
|
+
excluded: number;
|
|
746
|
+
/** Highest reward on an emitted gated row; `null` when none was emitted. */
|
|
747
|
+
maxEmittedReward: number | null;
|
|
748
|
+
/**
|
|
749
|
+
* The largest positive number found in the reward-DERIVED payload of an
|
|
750
|
+
* emitted gated row, and its path; `null` when there is none.
|
|
751
|
+
*
|
|
752
|
+
* This column exists because the release once certified CLEAN while leaking.
|
|
753
|
+
* `assertGateReport` inspected `outcome.reward` alone, so a gated row shipping
|
|
754
|
+
* `reward: 0` next to `metrics['layer.tests']: 1` — the deterministic verifier
|
|
755
|
+
* score the reward was computed from, and the per-rubric score dict of the
|
|
756
|
+
* Prime Intellect verifiers format — passed, and the card rendered "max reward
|
|
757
|
+
* | 0" over a file that carried the gamed signal at full value. A wrong
|
|
758
|
+
* certification is worse than the leak: it is the leak plus a document saying
|
|
759
|
+
* there isn't one.
|
|
760
|
+
*/
|
|
761
|
+
maxEmittedEvidence: EmittedEvidence | null;
|
|
762
|
+
/**
|
|
763
|
+
* Rows this format wrote carrying a positive reward whose producer DECLARED
|
|
764
|
+
* that no authenticity screen ever ran on it (`realness_screened: false`).
|
|
765
|
+
*
|
|
766
|
+
* Measured over EVERY emitted row, not just the gated ones: an unscreened
|
|
767
|
+
* reward is by definition one the gate never had a verdict on, so it is not in
|
|
768
|
+
* the gated set and a measurement scoped to that set would report 0 forever.
|
|
769
|
+
* `assertMinted` already refuses these, which is exactly why the release still
|
|
770
|
+
* measures them — the last door before a public dataset does not get to assume
|
|
771
|
+
* the earlier doors held.
|
|
772
|
+
*/
|
|
773
|
+
unscreenedPositiveRows: number;
|
|
774
|
+
/** Highest reward on such a row; `null` when there is none. */
|
|
775
|
+
maxUnscreenedReward: number | null;
|
|
776
|
+
/**
|
|
777
|
+
* The largest positive number found in an emitted gated row's UNDECLARED
|
|
778
|
+
* per-step payload, and its path; `null` when there is none.
|
|
779
|
+
*
|
|
780
|
+
* The column exists because the gate read `outcome` and nothing else for
|
|
781
|
+
* three rounds, so a gated line shipping `steps: [{kind, name, reward: 0.86}]`
|
|
782
|
+
* certified clean — the release accounting agreed with the exporter that a
|
|
783
|
+
* per-step reward was not a reward.
|
|
784
|
+
*/
|
|
785
|
+
maxEmittedStepEvidence: EmittedEvidence | null;
|
|
786
|
+
}
|
|
787
|
+
interface GateReport {
|
|
788
|
+
/** Gated lines in the release input, after the split/proposer filters. */
|
|
789
|
+
gatedLines: number;
|
|
790
|
+
byFormat: Partial<Record<ReleaseFormat, FormatGateCounts>>;
|
|
791
|
+
}
|
|
792
|
+
/** Rollout ids of every gated line, the key the emitted rows are matched on. */
|
|
793
|
+
declare function gatedRolloutIds(lines: readonly MintedRolloutLine[]): Set<string>;
|
|
794
|
+
/**
|
|
795
|
+
* Row refs per format. Written as one adapter per format so that the knowledge
|
|
796
|
+
* of WHERE the id and reward live in each published shape sits next to the
|
|
797
|
+
* assertion that uses it — an exporter that moves either field breaks here
|
|
798
|
+
* rather than silently reporting zero gated rows.
|
|
799
|
+
*/
|
|
800
|
+
declare const releaseRowRefs: {
|
|
801
|
+
sft: (rows: readonly SftRow[]) => ReleaseRowRef[];
|
|
802
|
+
verifiers: (rows: readonly VerifiersRolloutOutput[]) => ReleaseRowRef[];
|
|
803
|
+
rft: (rows: readonly RftItem[]) => ReleaseRowRef[];
|
|
804
|
+
raw: (lines: readonly MintedRolloutLine[]) => ReleaseRowRef[];
|
|
805
|
+
};
|
|
806
|
+
/** Measure one format's gated rows from the refs of the rows about to be written. */
|
|
807
|
+
declare function measureFormatGate(gated: ReadonlySet<string>, refs: readonly ReleaseRowRef[]): FormatGateCounts;
|
|
808
|
+
/**
|
|
809
|
+
* Fail the build when the measurement disagrees with the declared policy.
|
|
810
|
+
*
|
|
811
|
+
* Throws, never filters: an emitted positive reward on a gated row means an
|
|
812
|
+
* exporter upstream stopped applying the gate, and silently dropping the row
|
|
813
|
+
* would hide the producer that made it — the producer is the actual defect.
|
|
814
|
+
*
|
|
815
|
+
* Certifies the whole emitted outcome, not `reward` alone. The earlier version
|
|
816
|
+
* checked one field and therefore certified a release CLEAN while its
|
|
817
|
+
* `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in
|
|
818
|
+
* the top-level `metrics` dict — the card then rendered "max reward | 0" over
|
|
819
|
+
* exactly that file. A certification that is wrong is worse than an
|
|
820
|
+
* uncertified leak, so the checks it runs are no longer written down here at
|
|
821
|
+
* all: it iterates `GATE_CHECK_IDS` under its own declared policy.
|
|
822
|
+
*/
|
|
823
|
+
declare function assertGateReport(report: GateReport): void;
|
|
824
|
+
//#endregion
|
|
825
|
+
//#region src/rollout/release/scrub.d.ts
|
|
826
|
+
interface ScrubRule {
|
|
827
|
+
name: string;
|
|
828
|
+
pattern: RegExp;
|
|
829
|
+
/** Rewrite for one match; `g1` is the first capture group when present. */
|
|
830
|
+
rewrite: (match: string, g1?: string) => string;
|
|
831
|
+
}
|
|
832
|
+
declare const SCRUB_RULES: readonly ScrubRule[];
|
|
833
|
+
/** Rule name → number of matches rewritten. Always carries every rule (0 is data). */
|
|
834
|
+
type ScrubCounts = Record<string, number>;
|
|
835
|
+
declare function emptyScrubCounts(): ScrubCounts;
|
|
836
|
+
declare function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts;
|
|
837
|
+
declare function scrubText(text: string, counts: ScrubCounts): string;
|
|
838
|
+
/**
|
|
839
|
+
* Scrub every string value in a line; structure and key order are preserved.
|
|
840
|
+
*
|
|
841
|
+
* `assertMinted` on the way out rather than a cast: scrubbing rebuilds the
|
|
842
|
+
* object, so the brand has to be re-earned, and re-validating proves the rules
|
|
843
|
+
* did not rewrite a field the schema constrains (`reward` is a number, not a
|
|
844
|
+
* string, so no rule should ever touch it — this is what checks that).
|
|
845
|
+
*/
|
|
846
|
+
declare function scrubRolloutLine(line: MintedRolloutLine, counts: ScrubCounts): MintedRolloutLine;
|
|
847
|
+
declare function scrubLines(lines: MintedRolloutLine[]): {
|
|
848
|
+
lines: MintedRolloutLine[];
|
|
849
|
+
counts: ScrubCounts;
|
|
850
|
+
};
|
|
851
|
+
/**
|
|
852
|
+
* A `RolloutScrubber` (text → text) applying the full rule set — the
|
|
853
|
+
* default hook to pass to `mintRolloutRows({ scrub })` so lines are
|
|
854
|
+
* scrubbed at mint time, before they ever reach a ledger file. Release
|
|
855
|
+
* builds re-run `scrubLines` regardless (idempotent), so double-scrubbing
|
|
856
|
+
* is safe and counted as zero.
|
|
857
|
+
*/
|
|
858
|
+
declare function defaultRolloutScrubber(text: string): string;
|
|
859
|
+
//#endregion
|
|
860
|
+
//#region src/rollout/release/card.d.ts
|
|
861
|
+
declare const RELEASE_FORMATS: readonly ['sft', 'verifiers', 'rft', 'raw'];
|
|
862
|
+
type ReleaseFormat = (typeof RELEASE_FORMATS)[number];
|
|
863
|
+
/** Format → data file path inside the dataset dir (train split only). */
|
|
864
|
+
declare const FORMAT_FILES: Record<ReleaseFormat, string>;
|
|
865
|
+
interface DatasetCardInputs {
|
|
866
|
+
/** Scrubbed, release-filtered lines (what actually ships). */
|
|
867
|
+
lines: MintedRolloutLine[];
|
|
868
|
+
formats: ReleaseFormat[];
|
|
869
|
+
includeProposers: boolean;
|
|
870
|
+
/** Source ledger basenames, for provenance. */
|
|
871
|
+
sourceFiles: string[];
|
|
872
|
+
scrubTotals: ScrubCounts;
|
|
873
|
+
excluded: {
|
|
874
|
+
proposers: number;
|
|
875
|
+
nonTrain: number;
|
|
876
|
+
};
|
|
877
|
+
formatCounts: Partial<Record<ReleaseFormat, number>>;
|
|
878
|
+
/**
|
|
879
|
+
* Per-format anti-Goodhart accounting MEASURED on the rows the build wrote.
|
|
880
|
+
* Required, not optional: the card's only statement about the gate is a
|
|
881
|
+
* render of these numbers, so a card cannot be produced without them and
|
|
882
|
+
* cannot drift from the data files it ships beside.
|
|
883
|
+
*/
|
|
884
|
+
gate: GateReport;
|
|
885
|
+
}
|
|
886
|
+
declare function buildDatasetCard(inputs: DatasetCardInputs): string;
|
|
887
|
+
//#endregion
|
|
888
|
+
//#region src/rollout/release/hf-dataset.d.ts
|
|
889
|
+
interface BuildOptions {
|
|
890
|
+
out: string;
|
|
891
|
+
formats: ReleaseFormat[];
|
|
892
|
+
includeProposers: boolean;
|
|
893
|
+
}
|
|
894
|
+
interface ScrubReport {
|
|
895
|
+
/** Input ledger path → rule → rewrite count (only shipped lines are scrubbed). */
|
|
896
|
+
files: Record<string, ScrubCounts>;
|
|
897
|
+
totals: ScrubCounts;
|
|
898
|
+
excluded: {
|
|
899
|
+
proposers: number;
|
|
900
|
+
nonTrain: number;
|
|
901
|
+
};
|
|
902
|
+
}
|
|
903
|
+
interface BuildSummary {
|
|
904
|
+
inputs: string[];
|
|
905
|
+
read: number;
|
|
906
|
+
kept: number;
|
|
907
|
+
scrub: ScrubReport;
|
|
908
|
+
formatCounts: Partial<Record<ReleaseFormat, number>>;
|
|
909
|
+
/** Per-format anti-Goodhart accounting, measured on the rows written. */
|
|
910
|
+
gate: GateReport;
|
|
911
|
+
files: string[];
|
|
912
|
+
}
|
|
913
|
+
declare function buildHfDataset(inputs: string[], options: BuildOptions): Promise<BuildSummary>;
|
|
914
|
+
declare function planPushCommand(repo: string, outDir: string): string[];
|
|
915
|
+
declare function pushDataset(repo: string, outDir: string): void;
|
|
916
|
+
interface RolloutReleaseCliArgs extends BuildOptions {
|
|
917
|
+
inputs: string[];
|
|
918
|
+
push: string | null;
|
|
919
|
+
}
|
|
920
|
+
declare const ROLLOUT_RELEASE_USAGE = "usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]";
|
|
921
|
+
declare function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs;
|
|
922
|
+
/** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
|
|
923
|
+
declare function runRolloutReleaseCli(argv: string[]): Promise<number>;
|
|
924
|
+
//#endregion
|
|
925
|
+
export { ScorePreference as $, toVerifiersRolloutOutput as $t, ReleaseRowRef as A, GATE_CHECK_IDS as At, readOpencodeSessionMessages as B, RealnessLabels as Bt, scrubRolloutLine as C, HarborToolCall as Ct, FormatGateCounts as D, toHarborTrajectories as Dt, FORMAT_GATE_DISPOSITION as E, relabelImportedSplit as Et, DEFAULT_OPENCODE_DB as F, GateCheckedOutcome as Ft, claudeProjectSlug as G, VerifiersRolloutOutput as Gt, ClaudeTranscriptRef as H, RftItem as Ht, OpencodeSessionRow as I, GateEntryPoint as It, MintRolloutOptions as J, toJsonl as Jt, findClaudeTranscripts as K, VerifiersTokenUsage as Kt, findOpencodeSessionById as L, GatePolicy as Lt, gatedRolloutIds as M, GateCheck as Mt, measureFormatGate as N, GateCheckDisposition as Nt, GateDisposition as O, toHarborTrajectory as Ot, releaseRowRefs as P, GateCheckId as Pt, ScoreOrigin as Q, toSftRows as Qt, findOpencodeSessionsByDirectory as R, gateErrors as Rt, scrubLines as S, HarborSubagentTrajectoryRef as St, EmittedEvidence as T, fromHarborTrajectory as Tt, ClaudeUsageTotals as U, SftExportOptions as Ut, ClaudeTranscript as V, RewardRow as Vt, DEFAULT_CLAUDE_PROJECTS_DIR as W, SftRow as Wt, RolloutScrubber as X, toRftItem as Xt, MintRolloutResult as Y, toRewardRows as Yt, mintRolloutRows as Z, toRftItems as Zt, ScrubCounts as _, HarborMetrics as _t, ScrubReport as a, trainingScore as at, defaultRolloutScrubber as b, HarborStep as bt, planPushCommand as c, readRolloutLedger as ct, DatasetCardInputs as d, FromHarborOptions as dt, toVerifiersRolloutOutputs as en, isRealnessGated as et, FORMAT_FILES as f, HARBOR_IMPORT_GAP as ft, SCRUB_RULES as g, HarborImageSource as gt, buildDatasetCard as h, HarborFinalMetrics as ht, RolloutReleaseCliArgs as i, trainingReward as it, assertGateReport as j, GATE_POLICIES as jt, GateReport as k, GATE_CHECKS as kt, pushDataset as l, writeRolloutLedger as lt, ReleaseFormat as m, HarborContentPart as mt, BuildSummary as n, observedSplitScore as nt, buildHfDataset as o, appendRolloutLines as ot, RELEASE_FORMATS as p, HarborAgent as pt, readClaudeTranscript as q, realnessLabels as qt, ROLLOUT_RELEASE_USAGE as r, scoreOrigin as rt, parseRolloutReleaseArgs as s, readRolloutJournal as st, BuildOptions as t, observedScore as tt, runRolloutReleaseCli as u, ATIF_SCHEMA_VERSION as ut, ScrubRule as v, HarborObservation as vt, scrubText as w, HarborTrajectory as wt, emptyScrubCounts as x, HarborStepSource as xt, addScrubCounts as y, HarborObservationResult as yt, openOpencodeDb as z, gatedEvidenceOf as zt };
|
|
926
|
+
//# sourceMappingURL=index-2JJSA6-r2.d.ts.map
|