@tangle-network/agent-eval 0.129.0 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
package/dist/rollout/index.d.ts
CHANGED
|
@@ -1,2087 +1,3 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
* `tangle.rollout.v1` — THE canonical rollout serialization, owned by
|
|
5
|
-
* agent-eval. One JSONL line per agent invocation (a solo eval run, a
|
|
6
|
-
* supervisor episode, a worker session, a proposer shot, a judge call, an
|
|
7
|
-
* analyst pass), labeled with its task/split coordinates and a single
|
|
8
|
-
* scalar reward, carrying the FULL message transcript inline.
|
|
9
|
-
*
|
|
10
|
-
* This schema is the reconciliation of two prior producers:
|
|
11
|
-
* - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
|
|
12
|
-
* provenance hashes, the realness gate travelling into the reward,
|
|
13
|
-
* trace-derived steps.
|
|
14
|
-
* - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
|
|
15
|
-
* role, task.split/rep, parent_rollout_id, policy provenance, capture
|
|
16
|
-
* provenance, inline canonical chat-with-tools messages.
|
|
17
|
-
* Where the two conflicted, RunRecord-derived semantics won; the wire
|
|
18
|
-
* field names follow the ledger (snake_case). See `docs/rollout.md` for
|
|
19
|
-
* the field-by-field decision table.
|
|
20
|
-
*
|
|
21
|
-
* Messages are inlined — never referenced — because every harness store a
|
|
22
|
-
* rollout can be recovered from is mutable or garbage-collected. A line
|
|
23
|
-
* must stay a complete training/eval example on its own.
|
|
24
|
-
*
|
|
25
|
-
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
26
|
-
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
27
|
-
* flag: a gated line must never export as a positive training example.
|
|
28
|
-
*
|
|
29
|
-
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
30
|
-
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
31
|
-
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
32
|
-
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
33
|
-
* training export. The relationship between the two IS the invariant, so it is
|
|
34
|
-
* checked where every other structural claim about a line is checked.
|
|
35
|
-
*
|
|
36
|
-
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
37
|
-
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
38
|
-
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
39
|
-
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
40
|
-
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
41
|
-
* minted line passes — and the reward-bearing components are relocated to
|
|
42
|
-
* `provenance.gated_evidence`, which no exporter projects.
|
|
43
|
-
*
|
|
44
|
-
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
45
|
-
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
46
|
-
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
47
|
-
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
48
|
-
* here without anyone editing this file, and a check deliberately skipped has to
|
|
49
|
-
* name itself there.
|
|
50
|
-
*/
|
|
51
|
-
declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
52
|
-
/** `agent` = a solo evaluation run (no multi-agent topology). */
|
|
53
|
-
type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst';
|
|
54
|
-
declare const ROLLOUT_ROLES: readonly RolloutRole[];
|
|
55
|
-
/** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
|
|
56
|
-
type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
|
|
57
|
-
declare const ROLLOUT_SPLITS: readonly RolloutSplit[];
|
|
58
|
-
/** Splits that may ship in training exports. Everything else is fail-closed excluded. */
|
|
59
|
-
declare const TRAINABLE_SPLITS: readonly RolloutSplit[];
|
|
60
|
-
declare function isTrainableSplit(split: RolloutSplit): boolean;
|
|
61
|
-
/** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
|
|
62
|
-
type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
|
|
63
|
-
declare const ROLLOUT_CAPTURES: readonly RolloutCapture[];
|
|
64
|
-
type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
|
|
65
|
-
declare const CHAT_ROLES: readonly ChatRole[];
|
|
66
|
-
interface ChatToolCall {
|
|
67
|
-
id: string;
|
|
68
|
-
type: 'function';
|
|
69
|
-
function: {
|
|
70
|
-
name: string;
|
|
71
|
-
/** JSON-encoded argument object, exactly as the model emitted it. */
|
|
72
|
-
arguments: string;
|
|
73
|
-
};
|
|
74
|
-
}
|
|
75
|
-
interface ChatMessage {
|
|
76
|
-
role: ChatRole;
|
|
77
|
-
content: string | null;
|
|
78
|
-
/** Reasoning/thinking channel where the harness captured it (full fidelity). */
|
|
79
|
-
reasoning_content?: string;
|
|
80
|
-
tool_calls?: ChatToolCall[];
|
|
81
|
-
/** Required on role:"tool" — the ChatToolCall this result answers. */
|
|
82
|
-
tool_call_id?: string;
|
|
83
|
-
name?: string;
|
|
84
|
-
/**
|
|
85
|
-
* Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
|
|
86
|
-
* from another trajectory's context, not produced by the agent on this line.
|
|
87
|
-
* The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
|
|
88
|
-
* on it teaches the model to author text it never authored, and credits this
|
|
89
|
-
* run for another one's work. Absent = false (authored here).
|
|
90
|
-
*/
|
|
91
|
-
is_copied_context?: boolean;
|
|
92
|
-
}
|
|
93
|
-
interface ToolDef {
|
|
94
|
-
type: 'function';
|
|
95
|
-
function: {
|
|
96
|
-
name: string;
|
|
97
|
-
description?: string;
|
|
98
|
-
parameters?: Record<string, unknown>;
|
|
99
|
-
};
|
|
100
|
-
}
|
|
101
|
-
/**
|
|
102
|
-
* Compact trace-span projection (llm/tool step) carried alongside the
|
|
103
|
-
* conversation when the line was minted from a trace. Optional: lines
|
|
104
|
-
* recovered from harness stores have no span structure.
|
|
105
|
-
*/
|
|
106
|
-
interface RolloutStep {
|
|
107
|
-
kind: string;
|
|
108
|
-
name: string;
|
|
109
|
-
/** llm: last-message summary · tool: stringified args. Scrubbed. */
|
|
110
|
-
input?: string;
|
|
111
|
-
/** llm: output text · tool: stringified result. Scrubbed. */
|
|
112
|
-
output?: string;
|
|
113
|
-
status?: 'ok' | 'error';
|
|
114
|
-
durationMs?: number;
|
|
115
|
-
/**
|
|
116
|
-
* LLM inferences this span represents. 0 = deterministic dispatch with no
|
|
117
|
-
* model call — distinct from absent, which means the producer did not track it.
|
|
118
|
-
*/
|
|
119
|
-
llm_call_count?: number;
|
|
120
|
-
/** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
|
|
121
|
-
prompt_token_ids?: number[];
|
|
122
|
-
/** Exact completion tokenization; aligns index-wise with `logprobs`. */
|
|
123
|
-
completion_token_ids?: number[];
|
|
124
|
-
/**
|
|
125
|
-
* Per-completion-token log probabilities under the sampling policy. Required
|
|
126
|
-
* for off-policy correction (importance weighting) when the rollout was
|
|
127
|
-
* generated by a policy other than the one being trained.
|
|
128
|
-
*/
|
|
129
|
-
logprobs?: number[];
|
|
130
|
-
}
|
|
131
|
-
interface RolloutTask {
|
|
132
|
-
/** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
|
|
133
|
-
suite: string;
|
|
134
|
-
instance_id: string;
|
|
135
|
-
split: RolloutSplit;
|
|
136
|
-
/** Sampling seed the campaign pinned; null = not recorded. */
|
|
137
|
-
seed: number | null;
|
|
138
|
-
/** Replicate index (0-based). */
|
|
139
|
-
rep: number;
|
|
140
|
-
}
|
|
141
|
-
interface RolloutPolicy {
|
|
142
|
-
/** Harness that drove the invocation (e.g. "opencode", "claude", "pi-loops"). */
|
|
143
|
-
harness: string | null;
|
|
144
|
-
harness_version: string | null;
|
|
145
|
-
model: string | null;
|
|
146
|
-
provider: string | null;
|
|
147
|
-
/** Commit of the agent profile / candidate under evaluation. */
|
|
148
|
-
profile_commit: string | null;
|
|
149
|
-
/** sha256 of the effective prompt (post-steering), when recorded. */
|
|
150
|
-
prompt_hash?: string | null;
|
|
151
|
-
/** sha256 of the effective run config, when recorded. */
|
|
152
|
-
config_hash?: string | null;
|
|
153
|
-
/** Canonical agent-profile cell identity, when the run carries one. */
|
|
154
|
-
agent_profile_cell_id?: string | null;
|
|
155
|
-
/** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */
|
|
156
|
-
sampling: Record<string, unknown> | null;
|
|
157
|
-
}
|
|
158
|
-
interface RolloutOutcome {
|
|
159
|
-
/**
|
|
160
|
-
* THE single scalar training signal — the official verdict.
|
|
161
|
-
* null = no verdict exists for this invocation (a labeled gap, never 0).
|
|
162
|
-
*/
|
|
163
|
-
reward: number | null;
|
|
164
|
-
/** Where the reward came from (judge id; "/inherited" = parent episode's). */
|
|
165
|
-
reward_source: string | null;
|
|
166
|
-
/** Raw judge verdict record, verbatim. */
|
|
167
|
-
verdict: unknown;
|
|
168
|
-
/** Everything that is NOT the scalar reward. */
|
|
169
|
-
metrics: Record<string, unknown>;
|
|
170
|
-
is_completed: boolean;
|
|
171
|
-
is_truncated: boolean;
|
|
172
|
-
error: string | null;
|
|
173
|
-
/**
|
|
174
|
-
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
|
|
175
|
-
* its success signal. `true` requires `reward` to be 0 or null — the
|
|
176
|
-
* validator rejects the line otherwise — and the line never qualifies for
|
|
177
|
-
* SFT. Required on the wire: a line that does not state the flag does not
|
|
178
|
-
* validate, so no producer can dodge the gate by omitting it.
|
|
179
|
-
*
|
|
180
|
-
* `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
|
|
181
|
-
* numbers the reward was computed from are relocated to
|
|
182
|
-
* `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
|
|
183
|
-
* why zeroing the scalar alone was not enough.
|
|
184
|
-
*/
|
|
185
|
-
realness_gated: boolean;
|
|
186
|
-
/**
|
|
187
|
-
* Whether an authenticity SCREEN ever RAN on this reward — a different claim
|
|
188
|
-
* from `realness_gated`, which is the screen's VERDICT.
|
|
189
|
-
*
|
|
190
|
-
* `realness_gated: false` reads as "we looked and nothing fired". A producer
|
|
191
|
-
* with no screen at all was emitting exactly that, so a never-screened reward
|
|
192
|
-
* was indistinguishable on the wire from a screened-clean one, and the whole
|
|
193
|
-
* anti-Goodhart apparatus silently treated the first as the second. The two
|
|
194
|
-
* claims are now separable:
|
|
195
|
-
*
|
|
196
|
-
* - `true` — a screen ran; `realness_gated` is its verdict.
|
|
197
|
-
* - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
|
|
198
|
-
* `assertMinted` REFUSES such a line when its reward is above
|
|
199
|
-
* zero: an unscreened positive reward is precisely the signal
|
|
200
|
-
* the gate exists to qualify, and nothing has qualified it.
|
|
201
|
-
* - absent — not stated. Pre-unification ledgers land here, as does a
|
|
202
|
-
* `RunRecord` carrying no `outcome.realness` at all. Absent is
|
|
203
|
-
* read as "unknown", never as `false` (which would refuse most
|
|
204
|
-
* of the existing corpus) and never as `true`.
|
|
205
|
-
*/
|
|
206
|
-
realness_screened?: boolean;
|
|
207
|
-
}
|
|
208
|
-
interface RolloutCostBlock {
|
|
209
|
-
usd: number | null;
|
|
210
|
-
tokens_in: number | null;
|
|
211
|
-
tokens_out: number | null;
|
|
212
|
-
tokens_reasoning: number | null;
|
|
213
|
-
cache_read: number | null;
|
|
214
|
-
cache_write: number | null;
|
|
215
|
-
wall_s: number | null;
|
|
216
|
-
/**
|
|
217
|
-
* Total LLM inferences across the invocation (ATIF `llm_call_count`,
|
|
218
|
-
* aggregated). Optional and additive: absent = not tracked, never 0.
|
|
219
|
-
*/
|
|
220
|
-
llm_call_count?: number | null;
|
|
221
|
-
}
|
|
222
|
-
interface RolloutArtifacts {
|
|
223
|
-
patch_path: string | null;
|
|
224
|
-
run_dir: string | null;
|
|
225
|
-
/** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
|
|
226
|
-
transcript_ref: string | null;
|
|
227
|
-
}
|
|
228
|
-
/**
|
|
229
|
-
* The reward-bearing half of a GATED line's outcome, moved off `outcome` and
|
|
230
|
-
* parked here verbatim. Diagnostics, never training input — see
|
|
231
|
-
* `gateGamedOutcome`.
|
|
232
|
-
*/
|
|
233
|
-
interface GatedEvidence {
|
|
234
|
-
/** `outcome.metrics` exactly as the producer measured it. */
|
|
235
|
-
metrics?: Record<string, unknown>;
|
|
236
|
-
/** `outcome.verdict` verbatim — the judge record that claimed the success. */
|
|
237
|
-
verdict?: unknown;
|
|
238
|
-
/**
|
|
239
|
-
* The per-step fields `tangle.rollout.v1` does not declare, parked here when
|
|
240
|
-
* the gate projected `steps[]` down to the schema's own key set.
|
|
241
|
-
*
|
|
242
|
-
* A per-step reward is training signal exactly like the scalar, and `steps`
|
|
243
|
-
* rides through `toRewardRows` verbatim — so a gated line was shipping its
|
|
244
|
-
* step-level credit assignment at full value beside a `reward` of 0.
|
|
245
|
-
*/
|
|
246
|
-
steps?: unknown;
|
|
247
|
-
}
|
|
248
|
-
interface RolloutProvenance {
|
|
249
|
-
captured_at: string;
|
|
250
|
-
capture: RolloutCapture;
|
|
251
|
-
/**
|
|
252
|
-
* Why this line is incomplete. Required when `messages` is empty (the
|
|
253
|
-
* transcript could not be recovered); also set by interchange importers to
|
|
254
|
-
* name a MISSING LABEL — an imported trajectory carries no verdict, so
|
|
255
|
-
* `outcome.reward` is null and this says why.
|
|
256
|
-
*/
|
|
257
|
-
gap?: string;
|
|
258
|
-
/**
|
|
259
|
-
* Present only on a realness-gated line: the outcome fields the gate
|
|
260
|
-
* relocated, kept so an auditor can still see WHY the run was gated and what
|
|
261
|
-
* it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
|
|
262
|
-
* reads `outcome` and none reads `provenance`.
|
|
263
|
-
*/
|
|
264
|
-
gated_evidence?: GatedEvidence;
|
|
265
|
-
}
|
|
266
|
-
interface RolloutLine {
|
|
267
|
-
schema: typeof ROLLOUT_SCHEMA;
|
|
268
|
-
rollout_id: string;
|
|
269
|
-
/** Spawning invocation within the same episode (worker → supervisor). */
|
|
270
|
-
parent_rollout_id: string | null;
|
|
271
|
-
run_id: string;
|
|
272
|
-
/** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */
|
|
273
|
-
experiment_id: string | null;
|
|
274
|
-
/** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */
|
|
275
|
-
candidate_id: string | null;
|
|
276
|
-
/** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */
|
|
277
|
-
generation: number | null;
|
|
278
|
-
/** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */
|
|
279
|
-
candidate_index: number | null;
|
|
280
|
-
role: RolloutRole;
|
|
281
|
-
task: RolloutTask;
|
|
282
|
-
policy: RolloutPolicy;
|
|
283
|
-
/** Full transcript, inline. [] = gap line (see provenance.gap). */
|
|
284
|
-
messages: ChatMessage[];
|
|
285
|
-
tool_defs: ToolDef[];
|
|
286
|
-
/** Trace-span projections, when minted from a trace. */
|
|
287
|
-
steps?: RolloutStep[];
|
|
288
|
-
outcome: RolloutOutcome;
|
|
289
|
-
cost: RolloutCostBlock;
|
|
290
|
-
artifacts: RolloutArtifacts;
|
|
291
|
-
provenance: RolloutProvenance;
|
|
292
|
-
}
|
|
293
|
-
/**
|
|
294
|
-
* THE anti-Goodhart gate, applied to the WHOLE outcome as a TRANSFORMATION.
|
|
295
|
-
*
|
|
296
|
-
* Two prior rounds enforced the gate as a CHECK ON ONE FIELD at N call sites,
|
|
297
|
-
* and each round the next reward-bearing field leaked. The one that shipped:
|
|
298
|
-
* `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`
|
|
299
|
-
* with no gate, so a gated run exported `reward: 0` (correct) while the
|
|
300
|
-
* deterministic per-layer scores that reward was COMPUTED FROM — the
|
|
301
|
-
* `layer.*` keys `rl/verifiable-reward.ts` calls the RL training signal —
|
|
302
|
-
* shipped at 1.0, in the top-level `metrics` dict of the Prime Intellect
|
|
303
|
-
* verifiers format, which IS that format's per-rubric score dict. `verdict`
|
|
304
|
-
* leaks the same way into `toRftItem`'s `reference.verdict`, where a grader
|
|
305
|
-
* author reads `resolved: true` off a run that faked it.
|
|
306
|
-
*
|
|
307
|
-
* So the rule is no longer "zero the field we remembered". It is: if the gate
|
|
308
|
-
* fired, the outcome that leaves here carries NOTHING positive that was derived
|
|
309
|
-
* from the reward, whichever field a present or future exporter decides to
|
|
310
|
-
* read. `reward` is already forced to 0 upstream (`trainingReward`) and
|
|
311
|
-
* REJECTED here if it is not; `metrics` and `verdict` are relocated.
|
|
312
|
-
*
|
|
313
|
-
* WHERE they go, and why relocation rather than deletion: zeroing destroys the
|
|
314
|
-
* audit trail that shows why the run was gated and what it claimed, which is
|
|
315
|
-
* the row an auditor most wants and the labeled example a gaming DETECTOR
|
|
316
|
-
* trains on. `provenance.gated_evidence` keeps every byte, at a path no
|
|
317
|
-
* training exporter reads — all four release configs and every `rl/exporters`
|
|
318
|
-
* shape project from `outcome`, `messages`, `cost` and `task`; none projects
|
|
319
|
-
* `provenance`. Auditability preserved, training signal removed, and a future
|
|
320
|
-
* exporter that reads a field nobody thought of is safe by construction because
|
|
321
|
-
* the field is empty rather than because the exporter remembered to check.
|
|
322
|
-
*
|
|
323
|
-
* Idempotent: a second application finds nothing left to move and returns the
|
|
324
|
-
* line unchanged, so re-minting a line read back off a ledger cannot clobber
|
|
325
|
-
* the evidence it already carries.
|
|
326
|
-
*/
|
|
327
|
-
declare function gateGamedOutcome(line: RolloutLine): RolloutLine;
|
|
328
|
-
declare function validateRolloutLine(value: unknown): string[];
|
|
329
|
-
declare function assertRolloutLine(value: unknown, context?: string): asserts value is RolloutLine;
|
|
330
|
-
declare function isRolloutLine(value: unknown): value is RolloutLine;
|
|
331
|
-
/**
|
|
332
|
-
* Phantom property. `declare const` means it exists only in the type system:
|
|
333
|
-
* nothing is written at runtime, so a branded line still serializes to exactly
|
|
334
|
-
* the same JSON as a plain one.
|
|
335
|
-
*/
|
|
336
|
-
declare const MINTED_ROLLOUT: unique symbol;
|
|
337
|
-
/**
|
|
338
|
-
* A minted outcome states the gate verdict — it is not allowed to stay silent —
|
|
339
|
-
* and, when that verdict is `true`, carries nothing else the reward was derived
|
|
340
|
-
* from (`gateGamedOutcome` has run).
|
|
341
|
-
*/
|
|
342
|
-
interface MintedRolloutOutcome extends RolloutOutcome {
|
|
343
|
-
realness_gated: boolean;
|
|
344
|
-
}
|
|
345
|
-
/**
|
|
346
|
-
* A `RolloutLine` whose reward has been checked against the anti-Goodhart
|
|
347
|
-
* invariant. The type every training-data exporter takes.
|
|
348
|
-
*
|
|
349
|
-
* Why a brand and not just the interface: `RolloutLine` is structural, so any
|
|
350
|
-
* hand-built object literal of the right shape IS one — which is how a line
|
|
351
|
-
* declaring `{reward: 0.95, realness_gated: true}` reached the exporters
|
|
352
|
-
* despite them "only accepting a minted line". The phantom symbol makes the
|
|
353
|
-
* type nominal: it cannot be produced by writing an object literal, only by
|
|
354
|
-
* `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
|
|
355
|
-
* validates every line off disk), or an explicit, greppable `assertMinted`.
|
|
356
|
-
*
|
|
357
|
-
* Belt and braces on purpose. The brand closes first-party call sites at
|
|
358
|
-
* COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
|
|
359
|
-
* files, foreign imports, JSON from another process) where types are absent.
|
|
360
|
-
* Neither alone is enough.
|
|
361
|
-
*
|
|
362
|
-
* Assignable to `RolloutLine` in one direction only: readers, analysis, and
|
|
363
|
-
* the ledger writer keep taking the plain type.
|
|
364
|
-
*/
|
|
365
|
-
type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
|
|
366
|
-
readonly [MINTED_ROLLOUT]: true;
|
|
367
|
-
outcome: MintedRolloutOutcome;
|
|
368
|
-
};
|
|
369
|
-
/**
|
|
370
|
-
* Promote a line to the type the training exporters accept, applying the
|
|
371
|
-
* anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch
|
|
372
|
-
* — grep `assertMinted` to enumerate every place a line enters the training
|
|
373
|
-
* path without coming from mint or a ledger.
|
|
374
|
-
*
|
|
375
|
-
* The gate runs HERE, once, rather than at each producer, because this is the
|
|
376
|
-
* single funnel every minted line passes: `mintRolloutRows` calls it,
|
|
377
|
-
* `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the
|
|
378
|
-
* way out of a release, and a hand-built line has no other door. One
|
|
379
|
-
* transformation at the funnel means an already-published ledger holding a
|
|
380
|
-
* gated line with populated `metrics` is RE-GATED when it is read, instead of
|
|
381
|
-
* being rejected (which would make every such artifact unreadable) or trusted
|
|
382
|
-
* (which is the leak). Three steps, in this order:
|
|
383
|
-
*
|
|
384
|
-
* 1. VALIDATE the schema.
|
|
385
|
-
* 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today
|
|
386
|
-
* the reward relationship (which stays a REJECTION: a caller claiming
|
|
387
|
-
* `{reward: 0.95, realness_gated: true}` is a producer defect and must fail
|
|
388
|
-
* loudly, since laundering it into `reward: 0` here would hide the
|
|
389
|
-
* producer) and a positive reward the producer declared it never screened.
|
|
390
|
-
* 3. TRANSFORM the one check that policy marks `repair` — relocate the
|
|
391
|
-
* reward's components off `outcome` (`gateGamedOutcome`), so no exporter
|
|
392
|
-
* can leak them whichever field it reads.
|
|
393
|
-
*
|
|
394
|
-
* Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced
|
|
395
|
-
* here the moment its disposition in that policy says so.
|
|
396
|
-
*
|
|
397
|
-
* Also normalizes the optional wire flag to an explicit boolean.
|
|
398
|
-
* `realness_gated` is absent on pre-unification ledgers and absent means "not
|
|
399
|
-
* flagged" per the schema, so filling it in states a claim the line was already
|
|
400
|
-
* making, and makes the flag readable on every published row instead of most of
|
|
401
|
-
* them. `realness_screened` is NOT filled in: absent means "unknown", and
|
|
402
|
-
* inventing either value there would be the same overclaim this round removed.
|
|
403
|
-
*/
|
|
404
|
-
declare function assertMinted(value: unknown, context?: string): MintedRolloutLine;
|
|
405
|
-
/** `assertMinted` over a batch, naming the offending index in the error. */
|
|
406
|
-
declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
|
|
407
|
-
|
|
408
|
-
/**
|
|
409
|
-
* Pure exporters over `tangle.rollout.v1` lines → the training-data shapes
|
|
410
|
-
* the improvement loops feed:
|
|
411
|
-
* - SFT chat JSONL (clean trainable successes, {messages, metadata})
|
|
412
|
-
* - reward rows (every scored line, success or failure, with steps)
|
|
413
|
-
* - Prime Intellect verifiers RolloutOutput (prompt/completion split + reward)
|
|
414
|
-
* - OpenAI RFT items (prompt turns + verdict reference fields)
|
|
415
|
-
*
|
|
416
|
-
* All exporters are pure functions of the lines — filtering (never train on
|
|
417
|
-
* holdout, reward thresholds, the realness gate) happens HERE, on inline
|
|
418
|
-
* labels, no joins.
|
|
419
|
-
*
|
|
420
|
-
* Every exporter takes `MintedRolloutLine[]`, not `RolloutLine[]`: the reward
|
|
421
|
-
* on a minted line has been checked against the anti-Goodhart invariant, and
|
|
422
|
-
* the brand is what stops a hand-built object literal claiming a positive
|
|
423
|
-
* reward on a gamed run from being handed to an exporter that copies it
|
|
424
|
-
* verbatim into training data.
|
|
425
|
-
*/
|
|
426
|
-
|
|
427
|
-
/**
|
|
428
|
-
* The gate's two claims, which travel TOGETHER on every emitted row.
|
|
429
|
-
*
|
|
430
|
-
* `realness_gated` alone is ambiguous, and the ambiguity is exploitable:
|
|
431
|
-
* `false` reads as "we screened it and nothing fired", so a producer that has no
|
|
432
|
-
* screen at all emitted rows indistinguishable from screened-clean ones, and
|
|
433
|
-
* every consumer of the published dataset read them as clean. The second field
|
|
434
|
-
* is what separates the two claims, and it only removes the ambiguity if it
|
|
435
|
-
* reaches the WIRE — for a round it existed on `RolloutOutcome` and on no
|
|
436
|
-
* exported row shape at all, which left the published rows exactly as ambiguous
|
|
437
|
-
* as before.
|
|
438
|
-
*
|
|
439
|
-
* So there is one helper and every row shape spreads it. A row that states one
|
|
440
|
-
* claim without the other is not constructible by copying the pattern, and
|
|
441
|
-
* `exporters.test.ts` walks every emitted shape to prove none does.
|
|
442
|
-
*/
|
|
443
|
-
interface RealnessLabels {
|
|
444
|
-
/** The screen's VERDICT: the run faked its success signal. */
|
|
445
|
-
realness_gated: boolean;
|
|
446
|
-
/**
|
|
447
|
-
* Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its
|
|
448
|
-
* verdict. `false` = the producer declares it has none. `null` = not stated
|
|
449
|
-
* (pre-unification producers), which is "unknown" and never "clean".
|
|
450
|
-
*/
|
|
451
|
-
realness_screened: boolean | null;
|
|
452
|
-
}
|
|
453
|
-
declare function realnessLabels(line: MintedRolloutLine): RealnessLabels;
|
|
454
|
-
interface TrainingExportOptions {
|
|
455
|
-
/** Include held-out evaluation data in training output. Default false. */
|
|
456
|
-
allowHeldOutTrainingData?: boolean;
|
|
457
|
-
/** Require reward to be strictly greater than this value. Default 0. */
|
|
458
|
-
minimumQualityExclusive?: number;
|
|
459
|
-
}
|
|
460
|
-
/**
|
|
461
|
-
* What a signed-signal exporter (verifiers, RFT) does with lines that are not
|
|
462
|
-
* clean trainable successes — realness-gated lines above all.
|
|
463
|
-
*
|
|
464
|
-
* - 'exclude' — the default, the same fail-closed policy as every
|
|
465
|
-
* other training export: positive, completed,
|
|
466
|
-
* non-gated rows on a trainable split.
|
|
467
|
-
* - 'zero-and-flag' — keep them, at their non-positive (or null) reward,
|
|
468
|
-
* with `RealnessLabels` on the row. The dataset release
|
|
469
|
-
* sets this per `FORMAT_GATE_DISPOSITION`: in these
|
|
470
|
-
* formats the reward is a signed learning signal, so a
|
|
471
|
-
* gamed trajectory at reward 0 is a correct negative,
|
|
472
|
-
* and dropping it would bias the negative population
|
|
473
|
-
* toward honest failures and leave a trainer no example
|
|
474
|
-
* of gaming being penalized. The split policy is NOT
|
|
475
|
-
* relaxed: held-out lines still need the named opt-in.
|
|
476
|
-
*
|
|
477
|
-
* SFT deliberately has no such option — an SFT row is an imitation target and
|
|
478
|
-
* a gamed trajectory must never appear in one at any weight.
|
|
479
|
-
*/
|
|
480
|
-
type GatedLineDisposition = 'exclude' | 'zero-and-flag';
|
|
481
|
-
interface SignedSignalExportOptions extends TrainingExportOptions {
|
|
482
|
-
/** Disposition for non-trainable lines. Default 'exclude'. */
|
|
483
|
-
gatedLines?: GatedLineDisposition;
|
|
484
|
-
}
|
|
485
|
-
type SftExportOptions = TrainingExportOptions;
|
|
486
|
-
interface SftRow {
|
|
487
|
-
messages: ChatMessage[];
|
|
488
|
-
metadata: {
|
|
489
|
-
rollout_id: string;
|
|
490
|
-
run_id: string;
|
|
491
|
-
candidate_id: string | null;
|
|
492
|
-
instance_id: string;
|
|
493
|
-
reward: number;
|
|
494
|
-
} & RealnessLabels;
|
|
495
|
-
}
|
|
496
|
-
/**
|
|
497
|
-
* Supervised fine-tune rows: the completed conversation of each qualifying
|
|
498
|
-
* line. Fail-closed filters: trainable split only (never holdout/canary),
|
|
499
|
-
* reward strictly above `minimumQualityExclusive` (default 0), realness-gated
|
|
500
|
-
* lines never qualify, gap lines carry no trainable content, and
|
|
501
|
-
* copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001
|
|
502
|
-
* rule 7 — see `ChatMessage.is_copied_context`).
|
|
503
|
-
*
|
|
504
|
-
* `realness_gated` is therefore always `false` on an emitted row. It is carried
|
|
505
|
-
* anyway: an SFT row is a pure imitation target, so the row states its realness
|
|
506
|
-
* claims instead of making the reader know the format's policy, and carrying
|
|
507
|
-
* both flags on all four shapes is what lets the release accounting measure
|
|
508
|
-
* every config with one rule rather than skipping the one whose row shape
|
|
509
|
-
* happened to omit the field.
|
|
510
|
-
*/
|
|
511
|
-
declare function toSftRows(lines: MintedRolloutLine[], options?: SftExportOptions): SftRow[];
|
|
512
|
-
interface RewardRow {
|
|
513
|
-
/** First user turn — the task prompt. */
|
|
514
|
-
prompt: string;
|
|
515
|
-
steps: RolloutStep[];
|
|
516
|
-
reward: number;
|
|
517
|
-
metadata: {
|
|
518
|
-
rollout_id: string;
|
|
519
|
-
run_id: string;
|
|
520
|
-
candidate_id: string | null;
|
|
521
|
-
instance_id: string;
|
|
522
|
-
split: RolloutSplit;
|
|
523
|
-
} & RealnessLabels;
|
|
524
|
-
}
|
|
525
|
-
/**
|
|
526
|
-
* Reward-labeled rows for completed, positive-quality training runs.
|
|
527
|
-
*/
|
|
528
|
-
declare function toRewardRows(lines: MintedRolloutLine[], options?: TrainingExportOptions): RewardRow[];
|
|
529
|
-
interface VerifiersTokenUsage {
|
|
530
|
-
input_tokens: number | null;
|
|
531
|
-
output_tokens: number | null;
|
|
532
|
-
reasoning_tokens: number | null;
|
|
533
|
-
cache_read_tokens: number | null;
|
|
534
|
-
cache_write_tokens: number | null;
|
|
535
|
-
}
|
|
536
|
-
interface VerifiersRolloutOutput {
|
|
537
|
-
/** Messages through the last turn BEFORE the first assistant turn. */
|
|
538
|
-
prompt: ChatMessage[];
|
|
539
|
-
/** The first assistant turn onward — what the policy produced. */
|
|
540
|
-
completion: ChatMessage[];
|
|
541
|
-
reward: number | null;
|
|
542
|
-
metrics: Record<string, unknown>;
|
|
543
|
-
tool_defs: ToolDef[];
|
|
544
|
-
token_usage: VerifiersTokenUsage;
|
|
545
|
-
info: {
|
|
546
|
-
task: RolloutLine['task'];
|
|
547
|
-
policy: RolloutLine['policy'];
|
|
548
|
-
rollout_id: string;
|
|
549
|
-
run_id: string;
|
|
550
|
-
experiment_id: string | null;
|
|
551
|
-
candidate_id: string | null;
|
|
552
|
-
generation: number | null;
|
|
553
|
-
candidate_index: number | null;
|
|
554
|
-
role: RolloutLine['role'];
|
|
555
|
-
} & RealnessLabels;
|
|
556
|
-
}
|
|
557
|
-
declare function toVerifiersRolloutOutput(line: MintedRolloutLine): VerifiersRolloutOutput;
|
|
558
|
-
declare function toVerifiersRolloutOutputs(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): VerifiersRolloutOutput[];
|
|
559
|
-
interface RftItem {
|
|
560
|
-
/** Prompt turns only — the graded completion is re-sampled during RFT. */
|
|
561
|
-
messages: ChatMessage[];
|
|
562
|
-
/** Verdict/label fields the grader references as item.reference.* */
|
|
563
|
-
reference: {
|
|
564
|
-
reward: number | null;
|
|
565
|
-
reward_source: string | null;
|
|
566
|
-
verdict: unknown;
|
|
567
|
-
instance_id: string;
|
|
568
|
-
suite: string;
|
|
569
|
-
split: RolloutSplit;
|
|
570
|
-
rollout_id: string;
|
|
571
|
-
} & RealnessLabels;
|
|
572
|
-
}
|
|
573
|
-
declare function toRftItem(line: MintedRolloutLine): RftItem;
|
|
574
|
-
/** RFT needs a real prompt: lines whose transcript starts with prompt turns. */
|
|
575
|
-
declare function toRftItems(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): RftItem[];
|
|
576
|
-
declare function toJsonl(rows: ReadonlyArray<unknown>): string;
|
|
577
|
-
|
|
578
|
-
/**
|
|
579
|
-
* THE canonical list of anti-Goodhart gate checks, plus the TOTAL policy every
|
|
580
|
-
* entry point has to declare over it.
|
|
581
|
-
*
|
|
582
|
-
* Four rounds of adversarial review found the same defect four times, and it
|
|
583
|
-
* was never the check itself: it was the COMPOSITION. `validateRolloutLine`
|
|
584
|
-
* composed one check, `assertMinted` composed two, `assertRewardGate` composed
|
|
585
|
-
* two of the three, `assertGateReport` composed its own pair — each by hand, in
|
|
586
|
-
* its own file. So a check added to the package applied wherever its author
|
|
587
|
-
* happened to remember, and the guard that forgot it looked exactly like the
|
|
588
|
-
* guard that didn't. The last leak was literally that: `assertRewardGate`
|
|
589
|
-
* composed `reward-relationship` + `gated-evidence` and not `unscreened-reward`,
|
|
590
|
-
* so a never-screened positive reward that `assertMinted` correctly REFUSED
|
|
591
|
-
* walked through all four waist exporters at full value.
|
|
592
|
-
*
|
|
593
|
-
* The fix is to make hand-composition impossible rather than to add a third
|
|
594
|
-
* call to the two places that had two:
|
|
595
|
-
*
|
|
596
|
-
* - `GATE_CHECKS` is a TOTAL map over `GateCheckId`. A new id with no
|
|
597
|
-
* implementation does not compile.
|
|
598
|
-
* - `GatePolicy` is a TOTAL map over `GateCheckId`. Every entry point
|
|
599
|
-
* declares one, so a new id makes EVERY entry point's policy a type error
|
|
600
|
-
* until it is wired. Wiring it means writing `enforced`, `repairedBy(...)`
|
|
601
|
-
* or `omittedBecause(...)` — and the last two force a written reason, so a
|
|
602
|
-
* silent gap is not expressible.
|
|
603
|
-
* - every check carries a `tripwire`: the minimal outcome it must refuse.
|
|
604
|
-
* `gate-checks.test.ts` feeds each tripwire to every entry point that
|
|
605
|
-
* declares `enforced` and requires a rejection, so wiring a check to the
|
|
606
|
-
* wrong disposition is a TEST failure even when it type-checks.
|
|
607
|
-
*
|
|
608
|
-
* Adding a check is therefore: append the id, write the check, and the compiler
|
|
609
|
-
* enumerates every place that has to decide about it.
|
|
610
|
-
*/
|
|
611
|
-
|
|
612
|
-
/**
|
|
613
|
-
* Every gate check in the package, in the order they are applied.
|
|
614
|
-
*
|
|
615
|
-
* Order is load-bearing only for which message a caller sees first: a line that
|
|
616
|
-
* trips two checks reports the earlier one, and `reward-relationship` is first
|
|
617
|
-
* because it is the invariant the other three protect.
|
|
618
|
-
*/
|
|
619
|
-
declare const GATE_CHECK_IDS: readonly ["reward-relationship", "gated-evidence", "undeclared-step-payload", "unscreened-reward"];
|
|
620
|
-
type GateCheckId = (typeof GATE_CHECK_IDS)[number];
|
|
621
|
-
/**
|
|
622
|
-
* An outcome as it reaches a check.
|
|
623
|
-
*
|
|
624
|
-
* Deliberately accepts a raw record as well as the typed shape: the checks are
|
|
625
|
-
* the RUNTIME half of the gate, and the callers they exist for — JSON off a
|
|
626
|
-
* ledger, a plain-JavaScript consumer of the published package — arrive with no
|
|
627
|
-
* types at all. `Partial` because a tripwire states only the fields it trips on.
|
|
628
|
-
*/
|
|
629
|
-
type GateCheckedOutcome = Partial<RolloutOutcome> | Readonly<Record<string, unknown>>;
|
|
630
|
-
/**
|
|
631
|
-
* What a gate check reads: the reward-bearing surface of ONE LINE.
|
|
632
|
-
*
|
|
633
|
-
* For three rounds the subject was the OUTCOME alone, and that assumption is
|
|
634
|
-
* what produced the next leak rather than any missing check: `steps[]` sits on
|
|
635
|
-
* the LINE, outside `outcome`, so a per-step reward on a gated line was read by
|
|
636
|
-
* no check at all while `toRewardRows` copied it out verbatim — through the
|
|
637
|
-
* MINTED door, not merely the raw one. Widening the subject is what makes
|
|
638
|
-
* "somewhere else on the line" a place the checks can see.
|
|
639
|
-
*
|
|
640
|
-
* `outcome` is REQUIRED, and that is the point: a bare `RolloutOutcome` is then
|
|
641
|
-
* not assignable to a subject, so every call site that used to pass one is a
|
|
642
|
-
* COMPILE error until it passes the line instead. A subject with an optional
|
|
643
|
-
* `outcome` would have let the old call sites keep compiling while silently
|
|
644
|
-
* checking nothing — the exact failure this module exists to make impossible.
|
|
645
|
-
*/
|
|
646
|
-
interface GateSubject {
|
|
647
|
-
outcome: GateCheckedOutcome;
|
|
648
|
-
/** The line's trajectory steps, when it carries any. */
|
|
649
|
-
steps?: unknown;
|
|
650
|
-
}
|
|
651
|
-
interface GateCheck {
|
|
652
|
-
id: GateCheckId;
|
|
653
|
-
/** One sentence: what this check refuses. */
|
|
654
|
-
refuses: string;
|
|
655
|
-
/** One dotted-path message per defect; `[]` when the line is clean. */
|
|
656
|
-
errors: (subject: GateSubject) => string[];
|
|
657
|
-
/**
|
|
658
|
-
* Every minimal subject that MUST trip `errors` — the executable form of
|
|
659
|
-
* `refuses`, and the reason a check cannot be added without being provable.
|
|
660
|
-
* The calibration test feeds each one to every entry point declaring
|
|
661
|
-
* `enforced`.
|
|
662
|
-
*
|
|
663
|
-
* A LIST rather than one case: a check that refuses two distinct populations
|
|
664
|
-
* (a positive reward AND a reward it cannot read as a number) proved able to
|
|
665
|
-
* hold for the first while silently passing the second, so each population
|
|
666
|
-
* states its own tripwire and each is exercised separately.
|
|
667
|
-
*/
|
|
668
|
-
tripwires: GateSubject[];
|
|
669
|
-
}
|
|
670
|
-
/**
|
|
671
|
-
* The reward-bearing outcome fields that are NOT the scalar: the numbers the
|
|
672
|
-
* reward was computed from, and the verdict record that claimed it.
|
|
673
|
-
*
|
|
674
|
-
* Returned as one block rather than filtered key-by-key. A key-name heuristic
|
|
675
|
-
* ("zero anything matching `layer.*` or `/score/`") is the same defect shape as
|
|
676
|
-
* the line-oriented regex the AST score guard replaced: it holds until someone
|
|
677
|
-
* names a metric `pass_fraction`, and the next reward-shaped key ships at full
|
|
678
|
-
* value. The producer's OWN classification — "this is the scalar, that is
|
|
679
|
-
* everything else" — is the only partition that cannot be out-guessed.
|
|
680
|
-
*/
|
|
681
|
-
declare function gatedEvidenceOf(subject: GateSubject): GatedEvidence | undefined;
|
|
682
|
-
/**
|
|
683
|
-
* The registry. Total over `GateCheckId`, so an id with no check does not
|
|
684
|
-
* compile, and `GATE_CHECK_IDS` stays the single enumeration everything
|
|
685
|
-
* iterates.
|
|
686
|
-
*/
|
|
687
|
-
declare const GATE_CHECKS: {
|
|
688
|
-
readonly [K in GateCheckId]: GateCheck;
|
|
689
|
-
};
|
|
690
|
-
/**
|
|
691
|
-
* What ONE entry point does about ONE check.
|
|
692
|
-
*
|
|
693
|
-
* `repair` and `omit` both carry a mandatory sentence, which is the mechanism
|
|
694
|
-
* that keeps a legitimate omission distinguishable from a forgotten one: you
|
|
695
|
-
* cannot skip a check without writing down why, and the reasons are readable
|
|
696
|
-
* side by side in `GATE_POLICIES`.
|
|
697
|
-
*/
|
|
698
|
-
type GateCheckDisposition = {
|
|
699
|
-
readonly kind: 'enforce';
|
|
700
|
-
}
|
|
701
|
-
/** Resolved by TRANSFORMING the line instead of rejecting it; `by` names the function. */
|
|
702
|
-
| {
|
|
703
|
-
readonly kind: 'repair';
|
|
704
|
-
readonly by: string;
|
|
705
|
-
}
|
|
706
|
-
/** Deliberately not applied here; `because` states the reason. */
|
|
707
|
-
| {
|
|
708
|
-
readonly kind: 'omit';
|
|
709
|
-
readonly because: string;
|
|
710
|
-
};
|
|
711
|
-
/** Total over `GateCheckId`: a new check makes every policy literal a type error. */
|
|
712
|
-
type GatePolicy = {
|
|
713
|
-
readonly [K in GateCheckId]: GateCheckDisposition;
|
|
714
|
-
};
|
|
715
|
-
/**
|
|
716
|
-
* Every entry point that decides about the gate, and what it decides.
|
|
717
|
-
*
|
|
718
|
-
* Read this as the package's gate policy in one screen. The four entry points
|
|
719
|
-
* are not interchangeable — a validator that rejects, a mint funnel that
|
|
720
|
-
* repairs, a runtime backstop for untyped callers, and a release certifier over
|
|
721
|
-
* emitted rows — and the dispositions say which is which.
|
|
722
|
-
*/
|
|
723
|
-
declare const GATE_POLICIES: {
|
|
724
|
-
/**
|
|
725
|
-
* The schema validator. Rejects the reward relationship and NOTHING ELSE, on
|
|
726
|
-
* purpose: it runs on every line read off disk, and the other two conditions
|
|
727
|
-
* describe artifacts that already exist.
|
|
728
|
-
*/
|
|
729
|
-
readonly validateRolloutLine: {
|
|
730
|
-
readonly 'reward-relationship': {
|
|
731
|
-
readonly kind: "enforce";
|
|
732
|
-
};
|
|
733
|
-
readonly 'gated-evidence': GateCheckDisposition;
|
|
734
|
-
readonly 'undeclared-step-payload': GateCheckDisposition;
|
|
735
|
-
readonly 'unscreened-reward': GateCheckDisposition;
|
|
736
|
-
};
|
|
737
|
-
/**
|
|
738
|
-
* The mint funnel — the single door every `MintedRolloutLine` passes. Validates
|
|
739
|
-
* first (so the reward relationship has already been rejected), then refuses
|
|
740
|
-
* what cannot be repaired, then repairs what can.
|
|
741
|
-
*/
|
|
742
|
-
readonly assertMinted: {
|
|
743
|
-
readonly 'reward-relationship': {
|
|
744
|
-
readonly kind: "enforce";
|
|
745
|
-
};
|
|
746
|
-
readonly 'gated-evidence': GateCheckDisposition;
|
|
747
|
-
readonly 'undeclared-step-payload': GateCheckDisposition;
|
|
748
|
-
readonly 'unscreened-reward': {
|
|
749
|
-
readonly kind: "enforce";
|
|
750
|
-
};
|
|
751
|
-
};
|
|
752
|
-
/**
|
|
753
|
-
* The runtime backstop, and the one entry point with no license to omit
|
|
754
|
-
* anything: it exists for callers the type system never saw (plain JavaScript
|
|
755
|
-
* handing an object literal to a published exporter), so a check it skips is a
|
|
756
|
-
* check that does not run at all for them. This is where the fourth leak was.
|
|
757
|
-
*/
|
|
758
|
-
readonly assertRewardGate: {
|
|
759
|
-
readonly 'reward-relationship': {
|
|
760
|
-
readonly kind: "enforce";
|
|
761
|
-
};
|
|
762
|
-
readonly 'gated-evidence': {
|
|
763
|
-
readonly kind: "enforce";
|
|
764
|
-
};
|
|
765
|
-
readonly 'undeclared-step-payload': {
|
|
766
|
-
readonly kind: "enforce";
|
|
767
|
-
};
|
|
768
|
-
readonly 'unscreened-reward': {
|
|
769
|
-
readonly kind: "enforce";
|
|
770
|
-
};
|
|
771
|
-
};
|
|
772
|
-
/**
|
|
773
|
-
* The release certifier. Same checks, measured over the rows a release is
|
|
774
|
-
* ABOUT TO WRITE rather than over one line's outcome — see `REPORT_MEASURES`
|
|
775
|
-
* in `release/gate-report.ts`, which is the second total map this policy
|
|
776
|
-
* drives.
|
|
777
|
-
*/
|
|
778
|
-
readonly assertGateReport: {
|
|
779
|
-
readonly 'reward-relationship': {
|
|
780
|
-
readonly kind: "enforce";
|
|
781
|
-
};
|
|
782
|
-
readonly 'gated-evidence': {
|
|
783
|
-
readonly kind: "enforce";
|
|
784
|
-
};
|
|
785
|
-
readonly 'undeclared-step-payload': {
|
|
786
|
-
readonly kind: "enforce";
|
|
787
|
-
};
|
|
788
|
-
readonly 'unscreened-reward': {
|
|
789
|
-
readonly kind: "enforce";
|
|
790
|
-
};
|
|
791
|
-
};
|
|
792
|
-
};
|
|
793
|
-
/** Every entry point that declares a gate policy. */
|
|
794
|
-
type GateEntryPoint = keyof typeof GATE_POLICIES;
|
|
795
|
-
/**
|
|
796
|
-
* Run the checks one entry point enforces. The ONLY way an entry point should
|
|
797
|
-
* obtain gate errors — hand-composing two of the three is the bug this module
|
|
798
|
-
* exists to remove.
|
|
799
|
-
*/
|
|
800
|
-
declare function gateErrors(subject: GateSubject, policy: GatePolicy): string[];
|
|
801
|
-
|
|
802
|
-
/**
|
|
803
|
-
* Harbor ATIF-v1.7 interchange — `tangle.rollout.v1` ⇄ Agent Trajectory
|
|
804
|
-
* Interchange Format.
|
|
805
|
-
*
|
|
806
|
-
* ATIF is the portability format (spec:
|
|
807
|
-
* https://www.harborframework.com/docs/agents/trajectory-format, normative
|
|
808
|
-
* RFC: harbor-framework/harbor `rfcs/0001-trajectory-format.md`). It sits
|
|
809
|
-
* BELOW the waist of the rollout hourglass in both directions — export reads
|
|
810
|
-
* `RolloutLine[]`, import writes `RolloutLine[]` — and it is never a source
|
|
811
|
-
* of training labels:
|
|
812
|
-
*
|
|
813
|
-
* ATIF models NO reward, NO judge verdict, NO task/split coordinates.
|
|
814
|
-
*
|
|
815
|
-
* Consequences, both deliberate:
|
|
816
|
-
* - EXPORT drops `outcome.reward`, `outcome.reward_source` and
|
|
817
|
-
* `outcome.verdict` entirely. They are not smuggled into `extra`: a
|
|
818
|
-
* third-party reading our ATIF file must not be able to mistake an
|
|
819
|
-
* agent-eval judge score for something ATIF sanctioned.
|
|
820
|
-
* - IMPORT therefore mints UNLABELED lines: `reward: null` (the existing
|
|
821
|
-
* "null reward is a labeled gap, never 0" semantics), `verdict: null`,
|
|
822
|
-
* and a `provenance.gap` naming the missing label. An imported
|
|
823
|
-
* trajectory is not a training example until a judge scores it.
|
|
824
|
-
*
|
|
825
|
-
* Everything else we own that ATIF has no field for travels in a namespaced
|
|
826
|
-
* escrow at `extra.tangle.*`, so our own round-trip is exact while a foreign
|
|
827
|
-
* reader can ignore it. Fields that neither ATIF nor the escrow can carry
|
|
828
|
-
* come back explicitly null / fail-closed, never invented.
|
|
829
|
-
*
|
|
830
|
-
* THE ESCROW IS NAMESPACED, NOT AUTHENTICATED. Anyone can write
|
|
831
|
-
* `extra.tangle.*` into a file. So the escrow may restore what a value IS, but
|
|
832
|
-
* never what a line is ALLOWED to do: `task.split` is forced to `holdout` on
|
|
833
|
-
* every import regardless of what the document claims, and promoting an
|
|
834
|
-
* imported trajectory to a trainable split is an explicit, greppable act
|
|
835
|
-
* (`relabelImportedSplit`) rather than a property of the file. The document
|
|
836
|
-
* keeps its claim — the claim just is not authority.
|
|
837
|
-
*
|
|
838
|
-
* Multi-agent shape differs on purpose. ATIF EMBEDS children in
|
|
839
|
-
* `subagent_trajectories`; we keep a flat ledger with a normalized
|
|
840
|
-
* `parent_rollout_id` edge. Export assembles the tree, import flattens it.
|
|
841
|
-
* `session_id` is RUN-scoped in ATIF, so it carries `run_id` — the coordinate
|
|
842
|
-
* that is shared by every invocation of one run — not `rollout_id`, which
|
|
843
|
-
* identifies a single invocation and would split one run across session ids.
|
|
844
|
-
*
|
|
845
|
-
* ROUND-TRIPPING IS IDEMPOTENT: `import(export(import(export(x))))` is
|
|
846
|
-
* byte-identical to `import(export(x))`. Import composes `provenance.gap` as a
|
|
847
|
-
* de-duplicated ordered set rather than appending, and it emits every
|
|
848
|
-
* `ChatMessage` with keys in the canonical schema order (role, content,
|
|
849
|
-
* reasoning_content, tool_calls, tool_call_id, name, is_copied_context), so a
|
|
850
|
-
* ledger hashed on serialized bytes sees no diff across further passes. The
|
|
851
|
-
* FIRST import may re-order a producer's keys — that is the canonicalization.
|
|
852
|
-
*
|
|
853
|
-
* NOT building a Letta converter. Letta's trajectory-v1 is a strict subset of
|
|
854
|
-
* what we need from ATIF here — no per-step or aggregate cost, no
|
|
855
|
-
* multi-agent/subagent structure, no token-id or logprob channel — so a Letta
|
|
856
|
-
* sink would carry less than this one and add a second format to keep
|
|
857
|
-
* correct. Decision recorded in docs/rollout.md; do not re-litigate without a
|
|
858
|
-
* concrete consumer that reads Letta and cannot read ATIF.
|
|
859
|
-
*/
|
|
860
|
-
|
|
861
|
-
declare const ATIF_SCHEMA_VERSION = "ATIF-v1.7";
|
|
862
|
-
/** Gap note on every imported line — ATIF carries no verdict, so nothing is scored. */
|
|
863
|
-
declare const HARBOR_IMPORT_GAP = "imported from Harbor ATIF; no verdict";
|
|
864
|
-
type HarborStepSource = 'system' | 'user' | 'agent';
|
|
865
|
-
interface HarborImageSource {
|
|
866
|
-
media_type: string;
|
|
867
|
-
path: string;
|
|
868
|
-
}
|
|
869
|
-
interface HarborContentPart {
|
|
870
|
-
type: 'text' | 'image';
|
|
871
|
-
text?: string;
|
|
872
|
-
source?: HarborImageSource;
|
|
873
|
-
}
|
|
874
|
-
interface HarborToolCall {
|
|
875
|
-
tool_call_id: string;
|
|
876
|
-
function_name: string;
|
|
877
|
-
/** ATIF requires a decoded JSON object here, unlike our raw argument string. */
|
|
878
|
-
arguments: Record<string, unknown>;
|
|
879
|
-
extra?: Record<string, unknown>;
|
|
880
|
-
}
|
|
881
|
-
interface HarborSubagentTrajectoryRef {
|
|
882
|
-
trajectory_id?: string;
|
|
883
|
-
trajectory_path?: string;
|
|
884
|
-
/** Informational only since v1.7 — never a resolution key. */
|
|
885
|
-
session_id?: string;
|
|
886
|
-
extra?: Record<string, unknown>;
|
|
887
|
-
}
|
|
888
|
-
interface HarborObservationResult {
|
|
889
|
-
source_call_id?: string;
|
|
890
|
-
content?: string | HarborContentPart[];
|
|
891
|
-
subagent_trajectory_ref?: HarborSubagentTrajectoryRef[];
|
|
892
|
-
extra?: Record<string, unknown>;
|
|
893
|
-
}
|
|
894
|
-
interface HarborObservation {
|
|
895
|
-
results: HarborObservationResult[];
|
|
896
|
-
}
|
|
897
|
-
interface HarborMetrics {
|
|
898
|
-
prompt_tokens?: number;
|
|
899
|
-
completion_tokens?: number;
|
|
900
|
-
cached_tokens?: number;
|
|
901
|
-
cost_usd?: number;
|
|
902
|
-
prompt_token_ids?: number[];
|
|
903
|
-
completion_token_ids?: number[];
|
|
904
|
-
logprobs?: number[];
|
|
905
|
-
extra?: Record<string, unknown>;
|
|
906
|
-
}
|
|
907
|
-
interface HarborStep {
|
|
908
|
-
/** Ordinal, sequential from 1. */
|
|
909
|
-
step_id: number;
|
|
910
|
-
timestamp?: string;
|
|
911
|
-
source: HarborStepSource;
|
|
912
|
-
model_name?: string;
|
|
913
|
-
reasoning_effort?: string | number;
|
|
914
|
-
message: string | HarborContentPart[];
|
|
915
|
-
reasoning_content?: string;
|
|
916
|
-
tool_calls?: HarborToolCall[];
|
|
917
|
-
observation?: HarborObservation;
|
|
918
|
-
metrics?: HarborMetrics;
|
|
919
|
-
llm_call_count?: number;
|
|
920
|
-
is_copied_context?: boolean;
|
|
921
|
-
extra?: Record<string, unknown>;
|
|
922
|
-
}
|
|
923
|
-
interface HarborAgent {
|
|
924
|
-
name: string;
|
|
925
|
-
version: string;
|
|
926
|
-
model_name?: string;
|
|
927
|
-
/** OpenAI function-calling schema — byte-identical to our `ToolDef`. */
|
|
928
|
-
tool_definitions?: ToolDef[];
|
|
929
|
-
extra?: Record<string, unknown>;
|
|
930
|
-
}
|
|
931
|
-
interface HarborFinalMetrics {
|
|
932
|
-
total_prompt_tokens?: number;
|
|
933
|
-
total_completion_tokens?: number;
|
|
934
|
-
total_cached_tokens?: number;
|
|
935
|
-
total_cost_usd?: number;
|
|
936
|
-
total_steps?: number;
|
|
937
|
-
extra?: Record<string, unknown>;
|
|
938
|
-
}
|
|
939
|
-
interface HarborTrajectory {
|
|
940
|
-
schema_version: string;
|
|
941
|
-
session_id?: string;
|
|
942
|
-
/** Required on embedded subagents; we always set it so lines stay joinable. */
|
|
943
|
-
trajectory_id?: string;
|
|
944
|
-
agent: HarborAgent;
|
|
945
|
-
steps: HarborStep[];
|
|
946
|
-
notes?: string;
|
|
947
|
-
final_metrics?: HarborFinalMetrics;
|
|
948
|
-
continued_trajectory_ref?: string;
|
|
949
|
-
subagent_trajectories?: HarborTrajectory[];
|
|
950
|
-
extra?: Record<string, unknown>;
|
|
951
|
-
}
|
|
952
|
-
/**
|
|
953
|
-
* Assemble one episode's flat lines into a single ATIF trajectory tree,
|
|
954
|
-
* linked by `parent_rollout_id`.
|
|
955
|
-
*
|
|
956
|
-
* Reward, verdict and split are NOT emitted (ATIF models none of them); the
|
|
957
|
-
* split and the rest of the task coordinates survive only in `extra.tangle`.
|
|
958
|
-
*
|
|
959
|
-
* We deliberately do NOT synthesize an `observation.subagent_trajectory_ref`
|
|
960
|
-
* pointing at each child: our ledger records WHICH invocation spawned a
|
|
961
|
-
* worker, not which STEP did, and attaching the ref to a guessed step would
|
|
962
|
-
* fabricate a causal claim. Children are embedded in `subagent_trajectories`
|
|
963
|
-
* (each with the `trajectory_id` the spec requires) and the edge is stated in
|
|
964
|
-
* the child's escrowed `parent_rollout_id`.
|
|
965
|
-
*
|
|
966
|
-
* Throws when the lines are not one tree — use `toHarborTrajectories` for a forest.
|
|
967
|
-
*/
|
|
968
|
-
declare function toHarborTrajectory(lines: RolloutLine[]): HarborTrajectory;
|
|
969
|
-
/** Every independent tree in the input, one ATIF document each. */
|
|
970
|
-
declare function toHarborTrajectories(lines: RolloutLine[]): HarborTrajectory[];
|
|
971
|
-
interface FromHarborOptions {
|
|
972
|
-
/** Injected clock for deterministic output when the source carries no capture time. */
|
|
973
|
-
now?: () => Date;
|
|
974
|
-
}
|
|
975
|
-
/**
|
|
976
|
-
* Flatten an ATIF trajectory tree back into `tangle.rollout.v1` lines, parent
|
|
977
|
-
* first, each child carrying `parent_rollout_id`.
|
|
978
|
-
*
|
|
979
|
-
* Every line comes back UNLABELED: `reward`, `reward_source` and `verdict` are
|
|
980
|
-
* null and `provenance.gap` says why. ATIF models no verdict, so scoring an
|
|
981
|
-
* imported trajectory is a judge's job, not this function's. Every line lands
|
|
982
|
-
* on `holdout` whatever the document claims — see `relabelImportedSplit`.
|
|
983
|
-
*/
|
|
984
|
-
declare function fromHarborTrajectory(trajectory: HarborTrajectory, options?: FromHarborOptions): RolloutLine[];
|
|
985
|
-
/**
|
|
986
|
-
* THE explicit door out of `holdout` for imported lines.
|
|
987
|
-
*
|
|
988
|
-
* Import forces `holdout` because a document's own claim about its split is not
|
|
989
|
-
* evidence — anyone can write `extra.tangle.task.split`. Promoting a file to a
|
|
990
|
-
* trainable split is an operator's decision about provenance they verified, so
|
|
991
|
-
* it is a separate, greppable call: `grep relabelImportedSplit` enumerates
|
|
992
|
-
* every place foreign data was declared trainable, which is exactly the audit
|
|
993
|
-
* the trusted-escrow version made impossible.
|
|
994
|
-
*
|
|
995
|
-
* Returns plain `RolloutLine`s. They still have to pass `assertMinted` (and its
|
|
996
|
-
* anti-Goodhart check) to reach an exporter — re-labeling a split is not
|
|
997
|
-
* minting a reward.
|
|
998
|
-
*/
|
|
999
|
-
declare function relabelImportedSplit(lines: readonly RolloutLine[], split: RolloutSplit): RolloutLine[];
|
|
1000
|
-
|
|
1001
|
-
/**
|
|
1002
|
-
* Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`
|
|
1003
|
-
* lines. Writes validate BEFORE touching disk (a bad line never lands);
|
|
1004
|
-
* reads validate line-by-line and fail loud with the line number, because a
|
|
1005
|
-
* silently-skipped rollout is a corrupted dataset.
|
|
1006
|
-
*
|
|
1007
|
-
* "Validate" includes the anti-Goodhart invariant (a realness-gated line may
|
|
1008
|
-
* not carry a positive reward), so a poisoned line can neither enter a ledger
|
|
1009
|
-
* nor leave one.
|
|
1010
|
-
*
|
|
1011
|
-
* Two read modes, matching the two write-side row classes: `readRolloutLedger`
|
|
1012
|
-
* re-validates under the mint policy (training data), `readRolloutJournal`
|
|
1013
|
-
* under the write policy (supervision journals, whose unscreened positive
|
|
1014
|
-
* rewards are writable and must stay readable).
|
|
1015
|
-
*/
|
|
1016
|
-
|
|
1017
|
-
/** Replace the ledger file with exactly `lines`. */
|
|
1018
|
-
declare function writeRolloutLedger(path: string, lines: RolloutLine[]): Promise<void>;
|
|
1019
|
-
/** Append `lines` to the ledger file (created if absent). */
|
|
1020
|
-
declare function appendRolloutLines(path: string, lines: RolloutLine[]): Promise<void>;
|
|
1021
|
-
/**
|
|
1022
|
-
* Read and validate every line. Throws on the first malformed/invalid line
|
|
1023
|
-
* (with its 1-based line number) — fail-closed, never a silent drop.
|
|
1024
|
-
*
|
|
1025
|
-
* Validation includes the anti-Goodhart invariant, which is why the result is
|
|
1026
|
-
* `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this
|
|
1027
|
-
* process from outside the type system (another run, another machine, a
|
|
1028
|
-
* hand-edited JSONL), so this read is the runtime boundary where a poisoned
|
|
1029
|
-
* line is refused rather than exported.
|
|
1030
|
-
*/
|
|
1031
|
-
declare function readRolloutLedger(path: string): Promise<MintedRolloutLine[]>;
|
|
1032
|
-
/**
|
|
1033
|
-
* Read a ledger under the WRITE-side policy (`validateRolloutLine`), which
|
|
1034
|
-
* omits the unscreened-reward check. `writeRolloutLedger` accepts a
|
|
1035
|
-
* supervision-journal row (`realness_screened: false` with a positive reward
|
|
1036
|
-
* — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says
|
|
1037
|
-
* such rows "must stay writable, readable and reportable"; a read API that
|
|
1038
|
-
* only re-validated under `assertMinted` made every such file unreadable —
|
|
1039
|
-
* write-accepted but read-refused is a data-loss trap.
|
|
1040
|
-
*
|
|
1041
|
-
* The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here
|
|
1042
|
-
* can reach a training exporter without passing `assertMinted`, so the
|
|
1043
|
-
* promotion gate (which DOES enforce unscreened-reward) is exactly as closed
|
|
1044
|
-
* as before. Use `readRolloutLedger` when the file is training data.
|
|
1045
|
-
*/
|
|
1046
|
-
declare function readRolloutJournal(path: string): Promise<RolloutLine[]>;
|
|
1047
|
-
|
|
1048
|
-
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
1049
|
-
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
1050
|
-
interface AgentProfileSource {
|
|
1051
|
-
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
1052
|
-
kind: string;
|
|
1053
|
-
/** sha256 over the canonical source profile object. */
|
|
1054
|
-
hash: string;
|
|
1055
|
-
}
|
|
1056
|
-
interface AgentProfileHarness {
|
|
1057
|
-
id: string;
|
|
1058
|
-
version?: string;
|
|
1059
|
-
hash?: string;
|
|
1060
|
-
}
|
|
1061
|
-
interface AgentProfileCell {
|
|
1062
|
-
schemaVersion: AgentProfileCellSchemaVersion;
|
|
1063
|
-
cellId: string;
|
|
1064
|
-
profileId: string;
|
|
1065
|
-
sourceProfile: AgentProfileSource;
|
|
1066
|
-
harness?: AgentProfileHarness;
|
|
1067
|
-
model?: string;
|
|
1068
|
-
promptHash?: string;
|
|
1069
|
-
dimensions?: Record<string, AgentProfileDimensionValue>;
|
|
1070
|
-
}
|
|
1071
|
-
|
|
1072
|
-
type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
|
|
1073
|
-
interface BudgetSpec {
|
|
1074
|
-
tokens?: number;
|
|
1075
|
-
wallMs?: number;
|
|
1076
|
-
calls?: number;
|
|
1077
|
-
usd?: number;
|
|
1078
|
-
}
|
|
1079
|
-
interface RunOutcome$1 {
|
|
1080
|
-
score?: number;
|
|
1081
|
-
pass?: boolean;
|
|
1082
|
-
failureClass?: FailureClass;
|
|
1083
|
-
notes?: string;
|
|
1084
|
-
}
|
|
1085
|
-
/**
|
|
1086
|
-
* Layer — optional classification in a nested build workflow.
|
|
1087
|
-
* `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
|
|
1088
|
-
* `app-build`: sandbox harness that compiled + tested the generated scaffold.
|
|
1089
|
-
* `app-runtime`: a run of the generated agent against a domain scenario.
|
|
1090
|
-
* `meta`: any meta-eval (judge replay, correlation analysis).
|
|
1091
|
-
*/
|
|
1092
|
-
type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
|
|
1093
|
-
interface Run {
|
|
1094
|
-
runId: string;
|
|
1095
|
-
/**
|
|
1096
|
-
* Stable identifier of the scenario being executed.
|
|
1097
|
-
*
|
|
1098
|
-
* Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
|
|
1099
|
-
* input WITHOUT this field, substituting a sensible default
|
|
1100
|
-
* (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
|
|
1101
|
-
* curated scenario to anchor to (runtime / operator / meta-eval runs). This
|
|
1102
|
-
* keeps the persisted shape unambiguous for downstream filters + aggregations
|
|
1103
|
-
* while removing the boilerplate of inventing placeholder ids at the call site.
|
|
1104
|
-
*/
|
|
1105
|
-
scenarioId: string;
|
|
1106
|
-
variantId?: string;
|
|
1107
|
-
datasetVersion?: string;
|
|
1108
|
-
/** Git SHA of agent code at run time. */
|
|
1109
|
-
codeSha?: string;
|
|
1110
|
-
/** Hash of the prompt template + any system prompt. */
|
|
1111
|
-
promptSha?: string;
|
|
1112
|
-
/** Model id + date + system-prompt hash, concatenated. */
|
|
1113
|
-
modelFingerprint?: string;
|
|
1114
|
-
seed?: number;
|
|
1115
|
-
/** Arbitrary environment markers (shell, docker version, tz). */
|
|
1116
|
-
envFingerprint?: Record<string, string>;
|
|
1117
|
-
/** Version of the redaction rules applied to this run. */
|
|
1118
|
-
redactionVersion?: string;
|
|
1119
|
-
/** Parent run in a nested build workflow. A builder run's children are
|
|
1120
|
-
* app-build runs; those children are app-runtime runs. */
|
|
1121
|
-
parentRunId?: string;
|
|
1122
|
-
/** Stable project identifier — groups runs across chats + sessions. */
|
|
1123
|
-
projectId?: string;
|
|
1124
|
-
/** Chat/conversation identifier within a project. */
|
|
1125
|
-
chatId?: string;
|
|
1126
|
-
/** Layer classification — hint for aggregation; not enforced. */
|
|
1127
|
-
layer?: RunLayer;
|
|
1128
|
-
startedAt: number;
|
|
1129
|
-
endedAt?: number;
|
|
1130
|
-
status: RunStatus;
|
|
1131
|
-
outcome?: RunOutcome$1;
|
|
1132
|
-
budget?: BudgetSpec;
|
|
1133
|
-
/** Free-form labels for downstream grouping. */
|
|
1134
|
-
tags?: Record<string, string>;
|
|
1135
|
-
}
|
|
1136
|
-
type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
|
|
1137
|
-
type SpanStatus = 'ok' | 'error';
|
|
1138
|
-
interface SpanBase {
|
|
1139
|
-
spanId: string;
|
|
1140
|
-
parentSpanId?: string;
|
|
1141
|
-
runId: string;
|
|
1142
|
-
kind: SpanKind;
|
|
1143
|
-
name: string;
|
|
1144
|
-
startedAt: number;
|
|
1145
|
-
endedAt?: number;
|
|
1146
|
-
status?: SpanStatus;
|
|
1147
|
-
error?: string;
|
|
1148
|
-
/** Anything not covered by typed fields. Kept deliberately free-form. */
|
|
1149
|
-
attributes?: Record<string, unknown>;
|
|
1150
|
-
}
|
|
1151
|
-
interface Message {
|
|
1152
|
-
role: 'system' | 'user' | 'assistant' | 'tool';
|
|
1153
|
-
content: string;
|
|
1154
|
-
tokens?: number;
|
|
1155
|
-
/** Multi-modal content descriptors; blobs themselves live in Artifacts. */
|
|
1156
|
-
images?: Array<{
|
|
1157
|
-
artifactId?: string;
|
|
1158
|
-
url?: string;
|
|
1159
|
-
mime?: string;
|
|
1160
|
-
}>;
|
|
1161
|
-
}
|
|
1162
|
-
interface LlmSpan extends SpanBase {
|
|
1163
|
-
kind: 'llm';
|
|
1164
|
-
model: string;
|
|
1165
|
-
messages: Message[];
|
|
1166
|
-
output?: string;
|
|
1167
|
-
inputTokens?: number;
|
|
1168
|
-
/** All generated tokens, including the reasoning subset when present. */
|
|
1169
|
-
outputTokens?: number;
|
|
1170
|
-
cachedTokens?: number;
|
|
1171
|
-
cacheWriteTokens?: number;
|
|
1172
|
-
/** Reasoning-token subset of `outputTokens`. */
|
|
1173
|
-
reasoningTokens?: number;
|
|
1174
|
-
costUsd?: number;
|
|
1175
|
-
finishReason?: string;
|
|
1176
|
-
}
|
|
1177
|
-
interface ToolSpan extends SpanBase {
|
|
1178
|
-
kind: 'tool';
|
|
1179
|
-
toolName: string;
|
|
1180
|
-
args: unknown;
|
|
1181
|
-
/** False when the source observed the call but did not capture its arguments. */
|
|
1182
|
-
argsCaptured?: boolean;
|
|
1183
|
-
result?: unknown;
|
|
1184
|
-
latencyMs?: number;
|
|
1185
|
-
}
|
|
1186
|
-
interface RetrievalSpan extends SpanBase {
|
|
1187
|
-
kind: 'retrieval';
|
|
1188
|
-
query: string;
|
|
1189
|
-
hits: Array<{
|
|
1190
|
-
docId: string;
|
|
1191
|
-
score: number;
|
|
1192
|
-
content?: string;
|
|
1193
|
-
}>;
|
|
1194
|
-
}
|
|
1195
|
-
interface JudgeSpan extends SpanBase {
|
|
1196
|
-
kind: 'judge';
|
|
1197
|
-
judgeId: string;
|
|
1198
|
-
/** Span this judgment applies to. */
|
|
1199
|
-
targetSpanId: string;
|
|
1200
|
-
dimension: string;
|
|
1201
|
-
/** Numeric score (free-range; interpretation up to the judge). */
|
|
1202
|
-
score: number;
|
|
1203
|
-
rationale?: string;
|
|
1204
|
-
evidence?: string;
|
|
1205
|
-
}
|
|
1206
|
-
interface SandboxSpan extends SpanBase {
|
|
1207
|
-
kind: 'sandbox';
|
|
1208
|
-
image?: string;
|
|
1209
|
-
command?: string;
|
|
1210
|
-
exitCode?: number;
|
|
1211
|
-
testsTotal?: number;
|
|
1212
|
-
testsPassed?: number;
|
|
1213
|
-
stdoutHash?: string;
|
|
1214
|
-
stderrHash?: string;
|
|
1215
|
-
/** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
|
|
1216
|
-
wallMs?: number;
|
|
1217
|
-
}
|
|
1218
|
-
interface GenericSpan extends SpanBase {
|
|
1219
|
-
kind: 'agent' | 'custom';
|
|
1220
|
-
}
|
|
1221
|
-
type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
|
|
1222
|
-
type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
|
|
1223
|
-
interface TraceEvent {
|
|
1224
|
-
eventId: string;
|
|
1225
|
-
runId: string;
|
|
1226
|
-
spanId?: string;
|
|
1227
|
-
kind: EventKind;
|
|
1228
|
-
timestamp: number;
|
|
1229
|
-
payload: Record<string, unknown>;
|
|
1230
|
-
}
|
|
1231
|
-
interface BudgetLedgerEntry {
|
|
1232
|
-
runId: string;
|
|
1233
|
-
dimension: keyof BudgetSpec;
|
|
1234
|
-
limit: number;
|
|
1235
|
-
consumed: number;
|
|
1236
|
-
remaining: number;
|
|
1237
|
-
timestamp: number;
|
|
1238
|
-
breached: boolean;
|
|
1239
|
-
/** Span that triggered this entry, if any. */
|
|
1240
|
-
spanId?: string;
|
|
1241
|
-
}
|
|
1242
|
-
interface Artifact {
|
|
1243
|
-
artifactId: string;
|
|
1244
|
-
runId: string;
|
|
1245
|
-
spanId?: string;
|
|
1246
|
-
contentType: string;
|
|
1247
|
-
sizeBytes: number;
|
|
1248
|
-
/** sha256 in hex. */
|
|
1249
|
-
hash: string;
|
|
1250
|
-
/** External storage URL (R2, S3, filesystem path). */
|
|
1251
|
-
storageUrl?: string;
|
|
1252
|
-
/** Inline content for small blobs — keep under ~64KB. */
|
|
1253
|
-
inlineContent?: string;
|
|
1254
|
-
}
|
|
1255
|
-
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
1256
|
-
|
|
1257
|
-
/**
|
|
1258
|
-
* Paper-grade RunRecord schema + runtime validator.
|
|
1259
|
-
*
|
|
1260
|
-
* Every run that participates in a promotion gate, paper table, or
|
|
1261
|
-
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
1262
|
-
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
1263
|
-
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
1264
|
-
* holdout split tag. A task score is optional because execution-only records
|
|
1265
|
-
* must preserve missing labels instead of converting errors into zero quality.
|
|
1266
|
-
*
|
|
1267
|
-
* This is intentionally NOT a replacement for the rich `Run` /
|
|
1268
|
-
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
1269
|
-
* package. Those are runtime structures with full provenance. A
|
|
1270
|
-
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
1271
|
-
* row you'd put in a parquet file or paste into a notebook.
|
|
1272
|
-
*
|
|
1273
|
-
* Validate at the boundary:
|
|
1274
|
-
*
|
|
1275
|
-
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
1276
|
-
* const ok = isRunRecord(rawJson) // boolean check
|
|
1277
|
-
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
1278
|
-
*
|
|
1279
|
-
* The validator runs in pure TS — zod is intentionally NOT a
|
|
1280
|
-
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
1281
|
-
*/
|
|
1282
|
-
|
|
1283
|
-
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
1284
|
-
* combined train+test pool that the optimizer is allowed to read. */
|
|
1285
|
-
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
1286
|
-
/**
|
|
1287
|
-
* Explicit execution-lifecycle result for a run.
|
|
1288
|
-
*
|
|
1289
|
-
* This is separate from task quality (`outcome`) and failure classification.
|
|
1290
|
-
* Producers set it only from root-run or process evidence.
|
|
1291
|
-
*/
|
|
1292
|
-
type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
|
|
1293
|
-
interface RunTokenUsage {
|
|
1294
|
-
input: number;
|
|
1295
|
-
/** All generated tokens charged as output, including reasoning tokens. */
|
|
1296
|
-
output: number;
|
|
1297
|
-
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
1298
|
-
reasoning?: number;
|
|
1299
|
-
/** Prompt tokens served from a provider cache. */
|
|
1300
|
-
cached?: number;
|
|
1301
|
-
/** Prompt tokens written into a provider cache. */
|
|
1302
|
-
cacheWrite?: number;
|
|
1303
|
-
}
|
|
1304
|
-
/**
|
|
1305
|
-
* How a run's USD amount was obtained.
|
|
1306
|
-
*/
|
|
1307
|
-
type RunCostProvenance = {
|
|
1308
|
-
kind: 'observed';
|
|
1309
|
-
usd: number;
|
|
1310
|
-
} | {
|
|
1311
|
-
kind: 'estimated';
|
|
1312
|
-
usd: number;
|
|
1313
|
-
} | {
|
|
1314
|
-
kind: 'uncaptured';
|
|
1315
|
-
usd: null;
|
|
1316
|
-
};
|
|
1317
|
-
interface RunJudgeMetadata {
|
|
1318
|
-
model: string;
|
|
1319
|
-
promptVersion: string;
|
|
1320
|
-
/** [0,1] confidence the judge declared. Constant judge confidence
|
|
1321
|
-
* across many runs is a fallback signal (see `canary.ts`). */
|
|
1322
|
-
confidence: number;
|
|
1323
|
-
/** True if the judge degraded to a fallback path (rules-only,
|
|
1324
|
-
* prior-call cache, etc.). The canary uses this to alert. */
|
|
1325
|
-
fallback: boolean;
|
|
1326
|
-
}
|
|
1327
|
-
/**
|
|
1328
|
-
* Per-judge / per-dimension breakdown for runs scored by an ensemble of
|
|
1329
|
-
* judges over a multi-dimensional rubric.
|
|
1330
|
-
*
|
|
1331
|
-
* The collapsed `outcome.searchScore` / `holdoutScore` carries the
|
|
1332
|
-
* composite the gate uses. The full breakdown belongs here so consumers
|
|
1333
|
-
* can answer "which judge disagreed?", "which dimension dragged the
|
|
1334
|
-
* composite down?", and "did half the panel fail?" without re-running.
|
|
1335
|
-
*
|
|
1336
|
-
* `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
|
|
1337
|
-
* `composite` are convenience projections — derivable but precomputed so
|
|
1338
|
-
* downstream IRR primitives (`interRaterReliability`,
|
|
1339
|
-
* `corpusInterRaterAgreement`) and reporters don't pay the same
|
|
1340
|
-
* aggregation twice.
|
|
1341
|
-
*
|
|
1342
|
-
* Fail-loud discipline: judges that errored out land in `failedJudges`
|
|
1343
|
-
* by id. A missing key in `perJudge` is ambiguous (silent zero vs not
|
|
1344
|
-
* run); the explicit list makes a partial-failure recorded as such.
|
|
1345
|
-
*/
|
|
1346
|
-
interface JudgeScoresRecord {
|
|
1347
|
-
/** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
|
|
1348
|
-
perJudge: Record<string, Record<string, number>>;
|
|
1349
|
-
/** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
|
|
1350
|
-
perDimMean: Record<string, number>;
|
|
1351
|
-
/** Composite mean across successful judges. Mirrors the task score only
|
|
1352
|
-
* when `failedJudges` is empty. */
|
|
1353
|
-
composite: number;
|
|
1354
|
-
/** Judges that errored or returned an unparseable verdict. Recorded
|
|
1355
|
-
* by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
|
|
1356
|
-
* not inferred from missing keys in `perJudge`. */
|
|
1357
|
-
failedJudges?: string[];
|
|
1358
|
-
/** Free-form notes the judges emitted (joined across judges or
|
|
1359
|
-
* first-judge only — consumer's choice). */
|
|
1360
|
-
notes?: string;
|
|
1361
|
-
}
|
|
1362
|
-
interface RunOutcome {
|
|
1363
|
-
/** Score on the search/optimization split. Optional for holdout-only and
|
|
1364
|
-
* execution-only records. */
|
|
1365
|
-
searchScore?: number;
|
|
1366
|
-
/** Score on the held-out split. Optional for search-only and execution-only
|
|
1367
|
-
* records. When both scores are absent, the run is explicitly unlabeled. */
|
|
1368
|
-
holdoutScore?: number;
|
|
1369
|
-
/** Bag of any other metric the run produced — judge dimensions,
|
|
1370
|
-
* pass/fail counters, latency stats, etc. Numeric only — keeps
|
|
1371
|
-
* reporters honest. */
|
|
1372
|
-
raw: Record<string, number>;
|
|
1373
|
-
/** Per-judge / per-dim breakdown. Consumers writing ensemble
|
|
1374
|
-
* judgements populate this; substrate primitives like
|
|
1375
|
-
* `interRaterReliability` and `corpusInterRaterAgreement` accept
|
|
1376
|
-
* these records as input. Optional — single-judge or scalar-only
|
|
1377
|
-
* runs leave it unset. */
|
|
1378
|
-
judgeScores?: JudgeScoresRecord;
|
|
1379
|
-
/** Authenticity / realness verdict — did the run build the REAL thing on the
|
|
1380
|
-
* intended infra, or fake it (see `./authenticity`)? Optional: only domains
|
|
1381
|
-
* with an authenticity config populate it. Carried in the corpus so the
|
|
1382
|
-
* flywheel / off-policy learning can optimize for real completion, not gamed
|
|
1383
|
-
* pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
|
|
1384
|
-
* must not count as a real success regardless of `score`. */
|
|
1385
|
-
realness?: {
|
|
1386
|
-
score: number;
|
|
1387
|
-
gated: boolean;
|
|
1388
|
-
reason?: string;
|
|
1389
|
-
};
|
|
1390
|
-
}
|
|
1391
|
-
/**
|
|
1392
|
-
* Mandatory paper-grade fields for a single evaluation run. Optional
|
|
1393
|
-
* fields are extension points; mandatory fields throw if missing.
|
|
1394
|
-
*
|
|
1395
|
-
* Hash discipline:
|
|
1396
|
-
* - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
|
|
1397
|
-
* model (after any steering bundle merge).
|
|
1398
|
-
* - `configHash` is the sha256 of the effective run config (model,
|
|
1399
|
-
* temperature, tools, judges, splits). The pair (promptHash,
|
|
1400
|
-
* configHash) uniquely identifies an experiment cell.
|
|
1401
|
-
*
|
|
1402
|
-
* Model snapshot discipline:
|
|
1403
|
-
* - `model` MUST encode a snapshot version. Bare aliases like
|
|
1404
|
-
* `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
|
|
1405
|
-
* Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
|
|
1406
|
-
*/
|
|
1407
|
-
interface RunRecord {
|
|
1408
|
-
/** UUID for the run. */
|
|
1409
|
-
runId: string;
|
|
1410
|
-
/** Logical experiment grouping (a treatment vs a baseline within
|
|
1411
|
-
* the same sweep should share `experimentId`). */
|
|
1412
|
-
experimentId: string;
|
|
1413
|
-
/** Stable identifier for the candidate (variant) being run. The
|
|
1414
|
-
* promotion gate compares two `candidateId`s on matched items. */
|
|
1415
|
-
candidateId: string;
|
|
1416
|
-
/** RNG seed for the run. Always recorded — silent re-seeding is
|
|
1417
|
-
* the most common cause of non-reproducible numbers. */
|
|
1418
|
-
seed: number;
|
|
1419
|
-
/** Model identifier WITH snapshot version. */
|
|
1420
|
-
model: string;
|
|
1421
|
-
/** sha256 of the effective prompt (post-steering). */
|
|
1422
|
-
promptHash: string;
|
|
1423
|
-
/** sha256 of the effective config. */
|
|
1424
|
-
configHash: string;
|
|
1425
|
-
/** Git SHA the harness was run from. */
|
|
1426
|
-
commitSha: string;
|
|
1427
|
-
/** End-to-end wall-clock duration in milliseconds. */
|
|
1428
|
-
wallMs: number;
|
|
1429
|
-
/** Time spent queued before execution started, if known. */
|
|
1430
|
-
queueMs?: number;
|
|
1431
|
-
/** Total USD cost, or null when the producer could not capture one. */
|
|
1432
|
-
costUsd: number | null;
|
|
1433
|
-
/** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
|
|
1434
|
-
costProvenance: RunCostProvenance;
|
|
1435
|
-
/** Token usage breakdown. */
|
|
1436
|
-
tokenUsage: RunTokenUsage;
|
|
1437
|
-
/** Root-run or process terminal result. Never inferred from a child span. */
|
|
1438
|
-
terminalOutcome: RunTerminalOutcome;
|
|
1439
|
-
/** Root-run or process failure reason. Valid only for a failed, cancelled,
|
|
1440
|
-
* or incomplete terminal result; never populated from a child span. */
|
|
1441
|
-
terminalFailureReason?: string;
|
|
1442
|
-
/** Judge-side metadata, if a judge was used. */
|
|
1443
|
-
judgeMetadata?: RunJudgeMetadata;
|
|
1444
|
-
/** Per-split scores + raw bag. */
|
|
1445
|
-
outcome: RunOutcome;
|
|
1446
|
-
/** Canonical task-failure class drawn from the shared
|
|
1447
|
-
* `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
|
|
1448
|
-
* evidence. Execution errors belong in
|
|
1449
|
-
* `outcome.raw.execution_error_count`. */
|
|
1450
|
-
failureClass?: FailureClass;
|
|
1451
|
-
/** Free-form task-failure detail scoped under a non-success
|
|
1452
|
-
* `failureClass`. It is invalid without that class. */
|
|
1453
|
-
failureMode?: string;
|
|
1454
|
-
/** Which split this run was drawn from. */
|
|
1455
|
-
splitTag: RunSplitTag;
|
|
1456
|
-
/**
|
|
1457
|
-
* Stable scenario identifier the run observed or was scored against.
|
|
1458
|
-
* Comparison primitives match this identity rather than input order.
|
|
1459
|
-
*/
|
|
1460
|
-
scenarioId: string;
|
|
1461
|
-
/**
|
|
1462
|
-
* Canonical identity for the agent profile cell that produced this row:
|
|
1463
|
-
* profile artifact hash plus optional harness/model/prompt/reporting
|
|
1464
|
-
* dimensions. Use `agentProfile.cellId` to group persona sweeps and
|
|
1465
|
-
* longitudinal reports by the complete source profile, not by a loose
|
|
1466
|
-
* candidate label or opaque config hash.
|
|
1467
|
-
*/
|
|
1468
|
-
agentProfile?: AgentProfileCell;
|
|
1469
|
-
}
|
|
1470
|
-
|
|
1471
|
-
interface RunFilter {
|
|
1472
|
-
scenarioId?: string;
|
|
1473
|
-
variantId?: string;
|
|
1474
|
-
status?: RunStatus;
|
|
1475
|
-
since?: number;
|
|
1476
|
-
until?: number;
|
|
1477
|
-
tag?: {
|
|
1478
|
-
key: string;
|
|
1479
|
-
value: string;
|
|
1480
|
-
};
|
|
1481
|
-
parentRunId?: string;
|
|
1482
|
-
projectId?: string;
|
|
1483
|
-
chatId?: string;
|
|
1484
|
-
layer?: RunLayer;
|
|
1485
|
-
}
|
|
1486
|
-
interface SpanFilter {
|
|
1487
|
-
runId?: string;
|
|
1488
|
-
parentSpanId?: string;
|
|
1489
|
-
kind?: SpanKind;
|
|
1490
|
-
name?: string;
|
|
1491
|
-
toolName?: string;
|
|
1492
|
-
judgeId?: string;
|
|
1493
|
-
since?: number;
|
|
1494
|
-
until?: number;
|
|
1495
|
-
}
|
|
1496
|
-
interface EventFilter {
|
|
1497
|
-
runId?: string;
|
|
1498
|
-
spanId?: string;
|
|
1499
|
-
kind?: EventKind;
|
|
1500
|
-
since?: number;
|
|
1501
|
-
until?: number;
|
|
1502
|
-
}
|
|
1503
|
-
interface TraceStore {
|
|
1504
|
-
appendRun(run: Run): Promise<void>;
|
|
1505
|
-
updateRun(runId: string, patch: Partial<Run>): Promise<void>;
|
|
1506
|
-
appendSpan(span: Span): Promise<void>;
|
|
1507
|
-
updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
|
|
1508
|
-
appendEvent(event: TraceEvent): Promise<void>;
|
|
1509
|
-
appendArtifact(artifact: Artifact): Promise<void>;
|
|
1510
|
-
appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
|
|
1511
|
-
getRun(runId: string): Promise<Run | undefined>;
|
|
1512
|
-
listRuns(filter?: RunFilter): Promise<Run[]>;
|
|
1513
|
-
spans(filter?: SpanFilter): Promise<Span[]>;
|
|
1514
|
-
events(filter?: EventFilter): Promise<TraceEvent[]>;
|
|
1515
|
-
budget(runId: string): Promise<BudgetLedgerEntry[]>;
|
|
1516
|
-
artifacts(runId: string): Promise<Artifact[]>;
|
|
1517
|
-
}
|
|
1518
|
-
|
|
1519
|
-
/**
|
|
1520
|
-
* The two named score derivations every consumer must choose between.
|
|
1521
|
-
*
|
|
1522
|
-
* The anti-Goodhart gate (`outcome.realness.gated`) only holds if it is
|
|
1523
|
-
* impossible to read a run's score WITHOUT deciding whether the gate applies.
|
|
1524
|
-
* A bare `outcome.holdoutScore ?? outcome.searchScore` makes that decision
|
|
1525
|
-
* invisible — and silently answers "no gate", which is the wrong default on
|
|
1526
|
-
* every path that produces training data. So the expression lives here, once,
|
|
1527
|
-
* behind two names that force the caller to state the intent:
|
|
1528
|
-
*
|
|
1529
|
-
* - `trainingScore` / `trainingReward` — GATED. Anything that becomes
|
|
1530
|
-
* training data, or a reward a trainer consumes, uses these.
|
|
1531
|
-
* - `observedScore` — RAW. Analysis, reporting, and reward-hack DETECTION
|
|
1532
|
-
* need the ungated number; that is how a gamed run is visible at all.
|
|
1533
|
-
*
|
|
1534
|
-
* A leaf module on purpose: it imports only the `RunRecord` type, so gate and
|
|
1535
|
-
* reporting code can depend on it without pulling in the trace store that
|
|
1536
|
-
* `mint.ts` needs.
|
|
1537
|
-
*/
|
|
1538
|
-
|
|
1539
|
-
/**
|
|
1540
|
-
* Which split's score wins when a record carries both. `'holdout'` is the
|
|
1541
|
-
* canonical "real signal" default; `'search'` exists because some callers
|
|
1542
|
-
* deliberately score on the search split when both are present.
|
|
1543
|
-
*/
|
|
1544
|
-
type ScorePreference = 'holdout' | 'search';
|
|
1545
|
-
/** Only the outcome is read, so every accessor here accepts anything carrying one. */
|
|
1546
|
-
type Scored = Pick<RunRecord, 'outcome'>;
|
|
1547
|
-
/** True when the authenticity gate flagged the run as gamed (`realness.gated`). */
|
|
1548
|
-
declare function isRealnessGated(record: Scored): boolean;
|
|
1549
|
-
/**
|
|
1550
|
-
* The RAW score recorded on ONE split, with no cross-split fallback and no
|
|
1551
|
-
* anti-Goodhart gate.
|
|
1552
|
-
*
|
|
1553
|
-
* The narrowest of the three raw readers, and the one every split-scoped
|
|
1554
|
-
* consumer wants: a per-split report, a promotion gate, or a paired comparison
|
|
1555
|
-
* asks "what did this run score on the split I am summarising", and answering
|
|
1556
|
-
* it with the other split's number silently mixes populations. `undefined` =
|
|
1557
|
-
* that split was never scored.
|
|
1558
|
-
*
|
|
1559
|
-
* Same warning as `observedScore`: this INCLUDES runs flagged as gamed. Never
|
|
1560
|
-
* feed it into training data.
|
|
1561
|
-
*/
|
|
1562
|
-
declare function observedSplitScore(record: Scored, split: ScorePreference): number | undefined;
|
|
1563
|
-
/**
|
|
1564
|
-
* The RAW split score the run carries, with NO anti-Goodhart gate applied.
|
|
1565
|
-
*
|
|
1566
|
-
* INCLUDES RUNS FLAGGED AS GAMED (`outcome.realness.gated === true`); NEVER
|
|
1567
|
-
* feed this into training data — a fine-tune that sees it learns from gamed
|
|
1568
|
-
* successes. It is exported anyway because analysis, reporting, and
|
|
1569
|
-
* reward-hacking detection legitimately need the ungated number: forcing a
|
|
1570
|
-
* gamed run to 0 collapses the proxy signal toward ground truth and makes a
|
|
1571
|
-
* detector report "clean" on exactly the population that is being gamed.
|
|
1572
|
-
*
|
|
1573
|
-
* Returns `undefined` when the record carries neither score — an unscored run
|
|
1574
|
-
* is a labeled gap, not a measured zero, and each caller picks its own
|
|
1575
|
-
* sentinel (`?? 0`, `?? null`, skip, throw). Non-finite values are returned
|
|
1576
|
-
* as-is; callers that care keep their own `Number.isFinite` guard.
|
|
1577
|
-
*/
|
|
1578
|
-
declare function observedScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
1579
|
-
/** Which split actually carried the score, or that none did. */
|
|
1580
|
-
type ScoreOrigin = 'holdout' | 'search' | 'unscored';
|
|
1581
|
-
/**
|
|
1582
|
-
* Where `observedScore` / `trainingScore` read their number from — the
|
|
1583
|
-
* provenance label a rollout line's `reward_source` is built from, and the
|
|
1584
|
-
* only supported way to ask "was this run scored at all" without respelling
|
|
1585
|
-
* the field access.
|
|
1586
|
-
*/
|
|
1587
|
-
declare function scoreOrigin(record: Scored, prefer?: ScorePreference): ScoreOrigin;
|
|
1588
|
-
/**
|
|
1589
|
-
* The GATED score — the only derivation allowed to reach training data.
|
|
1590
|
-
*
|
|
1591
|
-
* A realness-gated run scores 0 no matter what it claims, so a fine-tune
|
|
1592
|
-
* cannot learn from a gamed success. An unscored run stays `undefined` (a
|
|
1593
|
-
* labeled gap), keeping "we never measured this" distinct from "we measured
|
|
1594
|
-
* zero"; callers that need a number apply their own sentinel.
|
|
1595
|
-
*/
|
|
1596
|
-
declare function trainingScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
1597
|
-
/**
|
|
1598
|
-
* `{reward, gated}` as written onto a minted `RolloutLine` — `trainingScore`
|
|
1599
|
-
* plus the flag itself, so the gate travels into the exported row and a
|
|
1600
|
-
* downstream filter can drop or down-weight the line.
|
|
1601
|
-
*
|
|
1602
|
-
* An unscored record yields `reward: null`, matching the schema's "no verdict
|
|
1603
|
-
* exists — a labeled gap, never 0" rule. It previously collapsed to 0, which
|
|
1604
|
-
* made a run nobody graded indistinguishable from one graded as a total
|
|
1605
|
-
* failure, and taught any trainer reading the row that the trajectory was bad.
|
|
1606
|
-
* A gated run still yields 0, because that IS a verdict: the gate decided.
|
|
1607
|
-
*/
|
|
1608
|
-
declare function trainingReward(record: Scored): {
|
|
1609
|
-
reward: number | null;
|
|
1610
|
-
gated: boolean;
|
|
1611
|
-
};
|
|
1612
|
-
|
|
1613
|
-
/**
|
|
1614
|
-
* Rollout minting — `tangle.rollout.v1` lines joined from the records the
|
|
1615
|
-
* substrate ALREADY keeps. There is no separate rollout store: a rollout
|
|
1616
|
-
* is the JOIN of a RunRecord (identity, provenance, cost, outcome) with
|
|
1617
|
-
* its trace (spans share `runId`), projected into the canonical line.
|
|
1618
|
-
*
|
|
1619
|
-
* Composition, not duplication:
|
|
1620
|
-
* - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)
|
|
1621
|
-
* - step structure → `buildTrajectory` over the shared TraceStore
|
|
1622
|
-
* - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
|
|
1623
|
-
* - PRM / reward-model → `reward-model-export.ts`
|
|
1624
|
-
*
|
|
1625
|
-
* Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is
|
|
1626
|
-
* never exported with a positive reward OR with any of the numbers that reward
|
|
1627
|
-
* was computed from. The gate travels into the training data (`reward` forced
|
|
1628
|
-
* to 0, `realness_gated: true`) and the whole outcome is transformed by
|
|
1629
|
-
* `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and
|
|
1630
|
-
* `verdict` to `provenance.gated_evidence`. Mint returns
|
|
1631
|
-
* `MintedRolloutLine[]`: the brand the training exporters require, which only
|
|
1632
|
-
* this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.
|
|
1633
|
-
*
|
|
1634
|
-
* A record carrying NEITHER split score is REJECTED (`ValidationError`), never
|
|
1635
|
-
* minted at 0 — "nobody graded this" is not the same claim as "graded a total
|
|
1636
|
-
* failure", and a trainer reading 0 learns the second. Lines that already
|
|
1637
|
-
* carry `reward: null` (interchange imports, existing ledgers) remain valid on
|
|
1638
|
-
* the wire; only the RunRecord→line door refuses.
|
|
1639
|
-
*
|
|
1640
|
-
* Records without spans become labeled GAP LINES (messages: [],
|
|
1641
|
-
* provenance.gap) — present in the output AND surfaced in
|
|
1642
|
-
* `missingTraces`; a capture gap is a finding, never a silent omission.
|
|
1643
|
-
*/
|
|
1644
|
-
|
|
1645
|
-
/** Redactor applied to every exported string (secrets, PII). Identity by default. */
|
|
1646
|
-
type RolloutScrubber = (text: string) => string;
|
|
1647
|
-
interface MintRolloutOptions {
|
|
1648
|
-
scrub?: RolloutScrubber;
|
|
1649
|
-
/** Cap steps per line (longest runs first drop middle steps). Default: no cap. */
|
|
1650
|
-
maxSteps?: number;
|
|
1651
|
-
/** Role recorded on every minted line. Default 'agent' (a solo eval run). */
|
|
1652
|
-
role?: RolloutRole;
|
|
1653
|
-
/** Task suite label. Default: the record's `experimentId`. */
|
|
1654
|
-
suite?: string;
|
|
1655
|
-
/** Injected clock for deterministic output. */
|
|
1656
|
-
now?: () => Date;
|
|
1657
|
-
}
|
|
1658
|
-
interface MintRolloutResult {
|
|
1659
|
-
rows: MintedRolloutLine[];
|
|
1660
|
-
/** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
|
|
1661
|
-
missingTraces: string[];
|
|
1662
|
-
}
|
|
1663
|
-
|
|
1664
|
-
/**
|
|
1665
|
-
* Join RunRecords with their traces into canonical rollout lines. Records
|
|
1666
|
-
* without spans are emitted as labeled gap lines and reported in
|
|
1667
|
-
* `missingTraces`. Execution-only records without a task score are rejected
|
|
1668
|
-
* because a missing training label is not a zero reward.
|
|
1669
|
-
*/
|
|
1670
|
-
declare function mintRolloutRows(records: RunRecord[], store: TraceStore, options?: MintRolloutOptions): Promise<MintRolloutResult>;
|
|
1671
|
-
|
|
1672
|
-
/**
|
|
1673
|
-
* Backfill reader over Claude Code project transcripts
|
|
1674
|
-
* (~/.claude/projects/<cwd-slug>/<sessionId>.jsonl) → canonical
|
|
1675
|
-
* chat-with-tools messages plus per-session token usage.
|
|
1676
|
-
*
|
|
1677
|
-
* Transcript lines consumed: type:"user" (string content or content blocks —
|
|
1678
|
-
* text + tool_result) and type:"assistant" (content blocks — thinking, text,
|
|
1679
|
-
* tool_use; message.usage carries tokens). Sidechain lines (isSidechain=true,
|
|
1680
|
-
* subagent threads) are separate invocations and are excluded from the main
|
|
1681
|
-
* transcript. Everything else (queue-operation, attachment, last-prompt…) is
|
|
1682
|
-
* transport metadata, not conversation.
|
|
1683
|
-
*/
|
|
1684
|
-
|
|
1685
|
-
declare const DEFAULT_CLAUDE_PROJECTS_DIR: string;
|
|
1686
|
-
/** Claude Code's project-directory slug for a working directory. */
|
|
1687
|
-
declare function claudeProjectSlug(cwd: string): string;
|
|
1688
|
-
interface ClaudeTranscriptRef {
|
|
1689
|
-
sessionId: string;
|
|
1690
|
-
path: string;
|
|
1691
|
-
}
|
|
1692
|
-
/** Transcript files recorded for sessions launched from `cwd`. */
|
|
1693
|
-
declare function findClaudeTranscripts(cwd: string, projectsDir?: string): Promise<ClaudeTranscriptRef[]>;
|
|
1694
|
-
interface ClaudeUsageTotals {
|
|
1695
|
-
tokensIn: number;
|
|
1696
|
-
tokensOut: number;
|
|
1697
|
-
cacheRead: number;
|
|
1698
|
-
cacheWrite: number;
|
|
1699
|
-
}
|
|
1700
|
-
interface ClaudeTranscript {
|
|
1701
|
-
messages: ChatMessage[];
|
|
1702
|
-
usage: ClaudeUsageTotals;
|
|
1703
|
-
/** Timestamp of the first conversation line; null = empty transcript. */
|
|
1704
|
-
startedAt: string | null;
|
|
1705
|
-
endedAt: string | null;
|
|
1706
|
-
model: string | null;
|
|
1707
|
-
}
|
|
1708
|
-
interface ReadClaudeTranscriptOptions {
|
|
1709
|
-
/**
|
|
1710
|
-
* Read the sidechain (subagent) thread instead of skipping it. Subagent
|
|
1711
|
-
* transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain
|
|
1712
|
-
* lines end to end, so their usage is invisible without this.
|
|
1713
|
-
*/
|
|
1714
|
-
readonly includeSidechain?: boolean;
|
|
1715
|
-
}
|
|
1716
|
-
/** Parse one transcript jsonl into canonical messages + usage totals. */
|
|
1717
|
-
declare function readClaudeTranscript(path: string, options?: ReadClaudeTranscriptOptions): Promise<ClaudeTranscript>;
|
|
1718
|
-
|
|
1719
|
-
/**
|
|
1720
|
-
* Read-only backfill reader over the opencode sqlite store
|
|
1721
|
-
* (~/.local/share/opencode/opencode.db) → canonical chat-with-tools messages.
|
|
1722
|
-
*
|
|
1723
|
-
* Schema consumed (observed, 2026-07): `session` rows carry directory /
|
|
1724
|
-
* parent_id / agent / model / cost / tokens_*; `message` rows carry a JSON
|
|
1725
|
-
* `data` blob ({role, modelID, providerID, tokens, cost, finish}); `part`
|
|
1726
|
-
* rows carry the actual content ({type: text|reasoning|tool|step-start|
|
|
1727
|
-
* step-finish|snapshot…}). Tool parts hold {callID, state:{input, output,
|
|
1728
|
-
* status}} — both the call and its result, which we split into an assistant
|
|
1729
|
-
* tool_call plus a role:"tool" result message.
|
|
1730
|
-
*
|
|
1731
|
-
* The store is mutable and can be corrupt (a `.corrupt-bak` sibling ships
|
|
1732
|
-
* next to it in the wild), so `openOpencodeDb` returns null instead of
|
|
1733
|
-
* throwing — callers record a gap line, never crash the backfill.
|
|
1734
|
-
*/
|
|
1735
|
-
|
|
1736
|
-
declare const DEFAULT_OPENCODE_DB: string;
|
|
1737
|
-
interface OpencodeSessionRow {
|
|
1738
|
-
id: string;
|
|
1739
|
-
parentId: string | null;
|
|
1740
|
-
directory: string;
|
|
1741
|
-
agent: string | null;
|
|
1742
|
-
/** Raw session.model JSON: {id, providerID, variant} where present. */
|
|
1743
|
-
model: {
|
|
1744
|
-
id?: string;
|
|
1745
|
-
providerID?: string;
|
|
1746
|
-
} | null;
|
|
1747
|
-
costUsd: number;
|
|
1748
|
-
tokensInput: number;
|
|
1749
|
-
tokensOutput: number;
|
|
1750
|
-
tokensReasoning: number;
|
|
1751
|
-
tokensCacheRead: number;
|
|
1752
|
-
tokensCacheWrite: number;
|
|
1753
|
-
timeCreated: number;
|
|
1754
|
-
timeUpdated: number;
|
|
1755
|
-
}
|
|
1756
|
-
/** Open the store read-only; null = unavailable/corrupt (caller records a gap). */
|
|
1757
|
-
declare function openOpencodeDb(path?: string): Promise<DatabaseSync | null>;
|
|
1758
|
-
/** Sessions whose cwd is `directory` (the worker-clone join key). */
|
|
1759
|
-
declare function findOpencodeSessionsByDirectory(db: DatabaseSync, directory: string): OpencodeSessionRow[];
|
|
1760
|
-
declare function findOpencodeSessionById(db: DatabaseSync, sessionId: string): OpencodeSessionRow | null;
|
|
1761
|
-
/**
|
|
1762
|
-
* Convert one session's message+part rows into canonical messages.
|
|
1763
|
-
* An opencode assistant message row spans several model steps; each step's
|
|
1764
|
-
* parts (reasoning → text → tool …) become one assistant message followed by
|
|
1765
|
-
* the role:"tool" results of its calls, preserving order.
|
|
1766
|
-
*/
|
|
1767
|
-
declare function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[];
|
|
1768
|
-
|
|
1769
|
-
/**
|
|
1770
|
-
* Per-format accounting of the anti-Goodhart gate for one dataset release.
|
|
1771
|
-
*
|
|
1772
|
-
* The defect this exists to make impossible: a dataset card that STATES what
|
|
1773
|
-
* the gate does while the build does something else. A sentence in a README is
|
|
1774
|
-
* a claim about bytes it never reads, so it drifts the moment an exporter
|
|
1775
|
-
* changes — and the drift ships to whoever downloads the dataset.
|
|
1776
|
-
*
|
|
1777
|
-
* So the card is not allowed to assert anything about the gate. The build
|
|
1778
|
-
* measures the rows it is ABOUT TO WRITE (`measureFormatGate`), the measurement
|
|
1779
|
-
* is checked against the declared per-format disposition (`assertGateReport`,
|
|
1780
|
-
* which throws rather than warns), and the card renders only numbers handed to
|
|
1781
|
-
* it. A card that disagrees with its own data files cannot be produced without
|
|
1782
|
-
* failing the build first.
|
|
1783
|
-
*
|
|
1784
|
-
* The dispositions themselves are the release policy, stated once as data:
|
|
1785
|
-
*
|
|
1786
|
-
* sft EXCLUDE — an SFT row is an imitation target. A gamed
|
|
1787
|
-
* trajectory must never be imitated, at any weight.
|
|
1788
|
-
* verifiers ZERO_AND_FLAG — reward is a signed learning signal here, so a
|
|
1789
|
-
* gamed trajectory at reward 0 is a correct
|
|
1790
|
-
* negative. Dropping it would bias the negative
|
|
1791
|
-
* population toward honest failures and leave a
|
|
1792
|
-
* trainer no example of what gaming looks like
|
|
1793
|
-
* when it is penalized.
|
|
1794
|
-
* rft ZERO_AND_FLAG — RFT re-samples the completion; only the prompt
|
|
1795
|
-
* and the grader's `reference.*` verdict ship, so
|
|
1796
|
-
* nothing gamed is imitated. The flag is what lets
|
|
1797
|
-
* a grader author skip the instance.
|
|
1798
|
-
* raw ZERO_AND_FLAG — a faithful audit dump. Removing rows from it
|
|
1799
|
-
* would defeat its only purpose, and the gated
|
|
1800
|
-
* row is the one an auditor most wants.
|
|
1801
|
-
*
|
|
1802
|
-
* `ZERO_AND_FLAG` is never `reward: 0` alone. Zeroing without the label makes a
|
|
1803
|
-
* faked success indistinguishable from an honest failure — it hides the gamed
|
|
1804
|
-
* population from the buyer instead of disclosing it. Every included format
|
|
1805
|
-
* carries `realness_gated` on the row itself.
|
|
1806
|
-
*
|
|
1807
|
-
* And `ZERO_AND_FLAG` means the whole outcome, not the scalar. A gated row that
|
|
1808
|
-
* ships `reward: 0` beside the per-layer verifier scores the reward was
|
|
1809
|
-
* computed from has not been zeroed in any sense a trainer respects; the
|
|
1810
|
-
* accounting therefore measures every reward-derived number each format writes,
|
|
1811
|
-
* not just the one field.
|
|
1812
|
-
*/
|
|
1813
|
-
|
|
1814
|
-
/** What a format does with a line the realness gate flagged. */
|
|
1815
|
-
type GateDisposition = 'exclude' | 'zero-and-flag';
|
|
1816
|
-
declare const FORMAT_GATE_DISPOSITION: Record<ReleaseFormat, GateDisposition>;
|
|
1817
|
-
/** What the gate accounting reads off an emitted row, per format. */
|
|
1818
|
-
interface ReleaseRowRef {
|
|
1819
|
-
rollout_id: string;
|
|
1820
|
-
reward: number | null;
|
|
1821
|
-
/**
|
|
1822
|
-
* The rest of the row that was DERIVED from the reward — the per-layer score
|
|
1823
|
-
* dict, the judge verdict record, whatever this format ships beside the
|
|
1824
|
-
* scalar. Walked for positive numbers, so the certification is about the
|
|
1825
|
-
* whole outcome rather than one field.
|
|
1826
|
-
*
|
|
1827
|
-
* Absent when the format's row carries nothing but the scalar. NOT the whole
|
|
1828
|
-
* row: `cost.tokens_in`, `wall_s` and `total_steps` are positive numbers that
|
|
1829
|
-
* have nothing to do with the reward, and a certification that flags them is
|
|
1830
|
-
* a certification nobody can act on.
|
|
1831
|
-
*/
|
|
1832
|
-
evidence?: unknown;
|
|
1833
|
-
/**
|
|
1834
|
-
* The screen claim AS EMITTED — read off the row, not off the line it came
|
|
1835
|
-
* from, because what ships is what matters. Required, not optional: an
|
|
1836
|
-
* optional field is how a format quietly opts out of the check that reads it,
|
|
1837
|
-
* and every emitted row shape carries `RealnessLabels` precisely so no adapter
|
|
1838
|
-
* has to.
|
|
1839
|
-
*/
|
|
1840
|
-
realness_screened: boolean | null;
|
|
1841
|
-
/**
|
|
1842
|
-
* The part of an emitted `steps[]` the wire format does not declare.
|
|
1843
|
-
*
|
|
1844
|
-
* Separate from `evidence` because the declared step fields are FULL of
|
|
1845
|
-
* legitimate positive numbers — `durationMs`, `llm_call_count`,
|
|
1846
|
-
* `prompt_token_ids` — and a certification that flags those is one nobody can
|
|
1847
|
-
* act on. Only the undeclared remainder is unclassified reward-bearing
|
|
1848
|
-
* payload, which is the same partition the check applies.
|
|
1849
|
-
*
|
|
1850
|
-
* Set only by formats whose row carries steps: today `raw` alone.
|
|
1851
|
-
*/
|
|
1852
|
-
stepEvidence?: unknown;
|
|
1853
|
-
}
|
|
1854
|
-
/** A positive number found inside an emitted gated row, with where it was. */
|
|
1855
|
-
interface EmittedEvidence {
|
|
1856
|
-
/** JSON-ish path from the row's evidence root, e.g. `metrics['layer.tests']`. */
|
|
1857
|
-
path: string;
|
|
1858
|
-
value: number;
|
|
1859
|
-
}
|
|
1860
|
-
interface FormatGateCounts {
|
|
1861
|
-
/** Gated lines that reached this format's exporter. */
|
|
1862
|
-
input: number;
|
|
1863
|
-
/** Gated rows the format actually wrote. */
|
|
1864
|
-
emitted: number;
|
|
1865
|
-
/**
|
|
1866
|
-
* Gated lines this format did not write. Not all of these are the gate:
|
|
1867
|
-
* `verifiers` also drops gap lines (empty transcript) and `rft` drops lines
|
|
1868
|
-
* with no prompt turn, so an excluded count can mix both causes.
|
|
1869
|
-
*/
|
|
1870
|
-
excluded: number;
|
|
1871
|
-
/** Highest reward on an emitted gated row; `null` when none was emitted. */
|
|
1872
|
-
maxEmittedReward: number | null;
|
|
1873
|
-
/**
|
|
1874
|
-
* The largest positive number found in the reward-DERIVED payload of an
|
|
1875
|
-
* emitted gated row, and its path; `null` when there is none.
|
|
1876
|
-
*
|
|
1877
|
-
* This column exists because the release once certified CLEAN while leaking.
|
|
1878
|
-
* `assertGateReport` inspected `outcome.reward` alone, so a gated row shipping
|
|
1879
|
-
* `reward: 0` next to `metrics['layer.tests']: 1` — the deterministic verifier
|
|
1880
|
-
* score the reward was computed from, and the per-rubric score dict of the
|
|
1881
|
-
* Prime Intellect verifiers format — passed, and the card rendered "max reward
|
|
1882
|
-
* | 0" over a file that carried the gamed signal at full value. A wrong
|
|
1883
|
-
* certification is worse than the leak: it is the leak plus a document saying
|
|
1884
|
-
* there isn't one.
|
|
1885
|
-
*/
|
|
1886
|
-
maxEmittedEvidence: EmittedEvidence | null;
|
|
1887
|
-
/**
|
|
1888
|
-
* Rows this format wrote carrying a positive reward whose producer DECLARED
|
|
1889
|
-
* that no authenticity screen ever ran on it (`realness_screened: false`).
|
|
1890
|
-
*
|
|
1891
|
-
* Measured over EVERY emitted row, not just the gated ones: an unscreened
|
|
1892
|
-
* reward is by definition one the gate never had a verdict on, so it is not in
|
|
1893
|
-
* the gated set and a measurement scoped to that set would report 0 forever.
|
|
1894
|
-
* `assertMinted` already refuses these, which is exactly why the release still
|
|
1895
|
-
* measures them — the last door before a public dataset does not get to assume
|
|
1896
|
-
* the earlier doors held.
|
|
1897
|
-
*/
|
|
1898
|
-
unscreenedPositiveRows: number;
|
|
1899
|
-
/** Highest reward on such a row; `null` when there is none. */
|
|
1900
|
-
maxUnscreenedReward: number | null;
|
|
1901
|
-
/**
|
|
1902
|
-
* The largest positive number found in an emitted gated row's UNDECLARED
|
|
1903
|
-
* per-step payload, and its path; `null` when there is none.
|
|
1904
|
-
*
|
|
1905
|
-
* The column exists because the gate read `outcome` and nothing else for
|
|
1906
|
-
* three rounds, so a gated line shipping `steps: [{kind, name, reward: 0.86}]`
|
|
1907
|
-
* certified clean — the release accounting agreed with the exporter that a
|
|
1908
|
-
* per-step reward was not a reward.
|
|
1909
|
-
*/
|
|
1910
|
-
maxEmittedStepEvidence: EmittedEvidence | null;
|
|
1911
|
-
}
|
|
1912
|
-
interface GateReport {
|
|
1913
|
-
/** Gated lines in the release input, after the split/proposer filters. */
|
|
1914
|
-
gatedLines: number;
|
|
1915
|
-
byFormat: Partial<Record<ReleaseFormat, FormatGateCounts>>;
|
|
1916
|
-
}
|
|
1917
|
-
/** Rollout ids of every gated line, the key the emitted rows are matched on. */
|
|
1918
|
-
declare function gatedRolloutIds(lines: readonly MintedRolloutLine[]): Set<string>;
|
|
1919
|
-
/**
|
|
1920
|
-
* Row refs per format. Written as one adapter per format so that the knowledge
|
|
1921
|
-
* of WHERE the id and reward live in each published shape sits next to the
|
|
1922
|
-
* assertion that uses it — an exporter that moves either field breaks here
|
|
1923
|
-
* rather than silently reporting zero gated rows.
|
|
1924
|
-
*/
|
|
1925
|
-
declare const releaseRowRefs: {
|
|
1926
|
-
sft: (rows: readonly SftRow[]) => ReleaseRowRef[];
|
|
1927
|
-
verifiers: (rows: readonly VerifiersRolloutOutput[]) => ReleaseRowRef[];
|
|
1928
|
-
rft: (rows: readonly RftItem[]) => ReleaseRowRef[];
|
|
1929
|
-
raw: (lines: readonly MintedRolloutLine[]) => ReleaseRowRef[];
|
|
1930
|
-
};
|
|
1931
|
-
/** Measure one format's gated rows from the refs of the rows about to be written. */
|
|
1932
|
-
declare function measureFormatGate(gated: ReadonlySet<string>, refs: readonly ReleaseRowRef[]): FormatGateCounts;
|
|
1933
|
-
/**
|
|
1934
|
-
* Fail the build when the measurement disagrees with the declared policy.
|
|
1935
|
-
*
|
|
1936
|
-
* Throws, never filters: an emitted positive reward on a gated row means an
|
|
1937
|
-
* exporter upstream stopped applying the gate, and silently dropping the row
|
|
1938
|
-
* would hide the producer that made it — the producer is the actual defect.
|
|
1939
|
-
*
|
|
1940
|
-
* Certifies the whole emitted outcome, not `reward` alone. The earlier version
|
|
1941
|
-
* checked one field and therefore certified a release CLEAN while its
|
|
1942
|
-
* `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in
|
|
1943
|
-
* the top-level `metrics` dict — the card then rendered "max reward | 0" over
|
|
1944
|
-
* exactly that file. A certification that is wrong is worse than an
|
|
1945
|
-
* uncertified leak, so the checks it runs are no longer written down here at
|
|
1946
|
-
* all: it iterates `GATE_CHECK_IDS` under its own declared policy.
|
|
1947
|
-
*/
|
|
1948
|
-
declare function assertGateReport(report: GateReport): void;
|
|
1949
|
-
|
|
1950
|
-
/**
|
|
1951
|
-
* Deterministic scrubbing pass over rollout-ledger lines before public release.
|
|
1952
|
-
*
|
|
1953
|
-
* Every rule is a pure regex rewrite applied to every string value in a line
|
|
1954
|
-
* (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed
|
|
1955
|
-
* line is still a valid `tangle.rollout.v1` line. Rules are idempotent:
|
|
1956
|
-
* scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is
|
|
1957
|
-
* the property the release pipeline relies on to prove nothing half-scrubbed
|
|
1958
|
-
* ships. Rule order matters: whole `KEY=value` env pairs are redacted before
|
|
1959
|
-
* the bare-key rule so one secret is never counted twice.
|
|
1960
|
-
*/
|
|
1961
|
-
|
|
1962
|
-
interface ScrubRule {
|
|
1963
|
-
name: string;
|
|
1964
|
-
pattern: RegExp;
|
|
1965
|
-
/** Rewrite for one match; `g1` is the first capture group when present. */
|
|
1966
|
-
rewrite: (match: string, g1?: string) => string;
|
|
1967
|
-
}
|
|
1968
|
-
declare const SCRUB_RULES: readonly ScrubRule[];
|
|
1969
|
-
/** Rule name → number of matches rewritten. Always carries every rule (0 is data). */
|
|
1970
|
-
type ScrubCounts = Record<string, number>;
|
|
1971
|
-
declare function emptyScrubCounts(): ScrubCounts;
|
|
1972
|
-
declare function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts;
|
|
1973
|
-
declare function scrubText(text: string, counts: ScrubCounts): string;
|
|
1974
|
-
/**
|
|
1975
|
-
* Scrub every string value in a line; structure and key order are preserved.
|
|
1976
|
-
*
|
|
1977
|
-
* `assertMinted` on the way out rather than a cast: scrubbing rebuilds the
|
|
1978
|
-
* object, so the brand has to be re-earned, and re-validating proves the rules
|
|
1979
|
-
* did not rewrite a field the schema constrains (`reward` is a number, not a
|
|
1980
|
-
* string, so no rule should ever touch it — this is what checks that).
|
|
1981
|
-
*/
|
|
1982
|
-
declare function scrubRolloutLine(line: MintedRolloutLine, counts: ScrubCounts): MintedRolloutLine;
|
|
1983
|
-
declare function scrubLines(lines: MintedRolloutLine[]): {
|
|
1984
|
-
lines: MintedRolloutLine[];
|
|
1985
|
-
counts: ScrubCounts;
|
|
1986
|
-
};
|
|
1987
|
-
/**
|
|
1988
|
-
* A `RolloutScrubber` (text → text) applying the full rule set — the
|
|
1989
|
-
* default hook to pass to `mintRolloutRows({ scrub })` so lines are
|
|
1990
|
-
* scrubbed at mint time, before they ever reach a ledger file. Release
|
|
1991
|
-
* builds re-run `scrubLines` regardless (idempotent), so double-scrubbing
|
|
1992
|
-
* is safe and counted as zero.
|
|
1993
|
-
*/
|
|
1994
|
-
declare function defaultRolloutScrubber(text: string): string;
|
|
1995
|
-
|
|
1996
|
-
/**
|
|
1997
|
-
* HuggingFace dataset-card (README.md) generation for a rollout-ledger release.
|
|
1998
|
-
*
|
|
1999
|
-
* The card is a pure function of the SCRUBBED lines plus the release options —
|
|
2000
|
-
* no timestamps, no environment reads — so rebuilding from the same ledger
|
|
2001
|
-
* yields byte-identical output. It documents the schema, provenance (run ids,
|
|
2002
|
-
* generations, the official judge), per-role reward semantics including the
|
|
2003
|
-
* inherited/contribution caveat, and a role × reward counts table.
|
|
2004
|
-
*/
|
|
2005
|
-
|
|
2006
|
-
declare const RELEASE_FORMATS: readonly ["sft", "verifiers", "rft", "raw"];
|
|
2007
|
-
type ReleaseFormat = (typeof RELEASE_FORMATS)[number];
|
|
2008
|
-
/** Format → data file path inside the dataset dir (train split only). */
|
|
2009
|
-
declare const FORMAT_FILES: Record<ReleaseFormat, string>;
|
|
2010
|
-
interface DatasetCardInputs {
|
|
2011
|
-
/** Scrubbed, release-filtered lines (what actually ships). */
|
|
2012
|
-
lines: MintedRolloutLine[];
|
|
2013
|
-
formats: ReleaseFormat[];
|
|
2014
|
-
includeProposers: boolean;
|
|
2015
|
-
/** Source ledger basenames, for provenance. */
|
|
2016
|
-
sourceFiles: string[];
|
|
2017
|
-
scrubTotals: ScrubCounts;
|
|
2018
|
-
excluded: {
|
|
2019
|
-
proposers: number;
|
|
2020
|
-
nonTrain: number;
|
|
2021
|
-
};
|
|
2022
|
-
formatCounts: Partial<Record<ReleaseFormat, number>>;
|
|
2023
|
-
/**
|
|
2024
|
-
* Per-format anti-Goodhart accounting MEASURED on the rows the build wrote.
|
|
2025
|
-
* Required, not optional: the card's only statement about the gate is a
|
|
2026
|
-
* render of these numbers, so a card cannot be produced without them and
|
|
2027
|
-
* cannot drift from the data files it ships beside.
|
|
2028
|
-
*/
|
|
2029
|
-
gate: GateReport;
|
|
2030
|
-
}
|
|
2031
|
-
declare function buildDatasetCard(inputs: DatasetCardInputs): string;
|
|
2032
|
-
|
|
2033
|
-
/**
|
|
2034
|
-
* One-command HuggingFace dataset release from rollout ledgers:
|
|
2035
|
-
*
|
|
2036
|
-
* agent-eval rollout-release <ledger.jsonl...> --out <dir> \
|
|
2037
|
-
* [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]
|
|
2038
|
-
*
|
|
2039
|
-
* Pipeline per input ledger: read + validate → fail-closed filters
|
|
2040
|
-
* (trainable split only; proposer sessions dropped unless
|
|
2041
|
-
* --include-proposers, they contain improvement-loop harness source) →
|
|
2042
|
-
* deterministic scrub → export the requested formats + scrub-report.json +
|
|
2043
|
-
* auto-generated README.md card. Deterministic: same inputs and flags →
|
|
2044
|
-
* byte-identical output dir.
|
|
2045
|
-
*
|
|
2046
|
-
* --push uploads the built dir with `huggingface-cli upload` only when the
|
|
2047
|
-
* CLI exists on PATH and HF_TOKEN is present in the env; the token is
|
|
2048
|
-
* never printed. Everything else runs fully offline.
|
|
2049
|
-
*/
|
|
2050
|
-
|
|
2051
|
-
interface BuildOptions {
|
|
2052
|
-
out: string;
|
|
2053
|
-
formats: ReleaseFormat[];
|
|
2054
|
-
includeProposers: boolean;
|
|
2055
|
-
}
|
|
2056
|
-
interface ScrubReport {
|
|
2057
|
-
/** Input ledger path → rule → rewrite count (only shipped lines are scrubbed). */
|
|
2058
|
-
files: Record<string, ScrubCounts>;
|
|
2059
|
-
totals: ScrubCounts;
|
|
2060
|
-
excluded: {
|
|
2061
|
-
proposers: number;
|
|
2062
|
-
nonTrain: number;
|
|
2063
|
-
};
|
|
2064
|
-
}
|
|
2065
|
-
interface BuildSummary {
|
|
2066
|
-
inputs: string[];
|
|
2067
|
-
read: number;
|
|
2068
|
-
kept: number;
|
|
2069
|
-
scrub: ScrubReport;
|
|
2070
|
-
formatCounts: Partial<Record<ReleaseFormat, number>>;
|
|
2071
|
-
/** Per-format anti-Goodhart accounting, measured on the rows written. */
|
|
2072
|
-
gate: GateReport;
|
|
2073
|
-
files: string[];
|
|
2074
|
-
}
|
|
2075
|
-
declare function buildHfDataset(inputs: string[], options: BuildOptions): Promise<BuildSummary>;
|
|
2076
|
-
declare function planPushCommand(repo: string, outDir: string): string[];
|
|
2077
|
-
declare function pushDataset(repo: string, outDir: string): void;
|
|
2078
|
-
interface RolloutReleaseCliArgs extends BuildOptions {
|
|
2079
|
-
inputs: string[];
|
|
2080
|
-
push: string | null;
|
|
2081
|
-
}
|
|
2082
|
-
declare const ROLLOUT_RELEASE_USAGE = "usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]";
|
|
2083
|
-
declare function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs;
|
|
2084
|
-
/** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
|
|
2085
|
-
declare function runRolloutReleaseCli(argv: string[]): Promise<number>;
|
|
2086
|
-
|
|
2087
|
-
export { ATIF_SCHEMA_VERSION, type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, type EmittedEvidence, FORMAT_FILES, FORMAT_GATE_DISPOSITION, type FormatGateCounts, type FromHarborOptions, GATE_CHECKS, GATE_CHECK_IDS, GATE_POLICIES, type GateCheck, type GateCheckDisposition, type GateCheckId, type GateCheckedOutcome, type GateDisposition, type GateEntryPoint, type GatePolicy, type GateReport, type GatedEvidence, HARBOR_IMPORT_GAP, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type RealnessLabels, type ReleaseFormat, type ReleaseRowRef, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScoreOrigin, type ScorePreference, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertGateReport, assertMinted, assertMintedLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, fromHarborTrajectory, gateErrors, gateGamedOutcome, gatedEvidenceOf, gatedRolloutIds, isRealnessGated, isRolloutLine, isTrainableSplit, measureFormatGate, mintRolloutRows, observedScore, observedSplitScore, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutJournal, readRolloutLedger, realnessLabels, relabelImportedSplit, releaseRowRefs, runRolloutReleaseCli, scoreOrigin, scrubLines, scrubRolloutLine, scrubText, toHarborTrajectories, toHarborTrajectory, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, trainingReward, trainingScore, validateRolloutLine, writeRolloutLedger };
|
|
1
|
+
import { A as isTrainableSplit, C as TRAINABLE_SPLITS, D as assertRolloutLine, E as assertMintedLines, O as gateGamedOutcome, S as RolloutTask, T as assertMinted, _ as RolloutPolicy, a as GatedEvidence, b as RolloutSplit, c as ROLLOUT_CAPTURES, d as ROLLOUT_SPLITS, f as RolloutArtifacts, g as RolloutOutcome, h as RolloutLine, i as ChatToolCall, j as validateRolloutLine, k as isRolloutLine, l as ROLLOUT_ROLES, m as RolloutCostBlock, n as ChatMessage, o as MintedRolloutLine, p as RolloutCapture, r as ChatRole, s as MintedRolloutOutcome, t as CHAT_ROLES, u as ROLLOUT_SCHEMA, v as RolloutProvenance, w as ToolDef, x as RolloutStep, y as RolloutRole } from "../schema-Cef2cFmb.js";
|
|
2
|
+
import { $ as ScorePreference, $t as toVerifiersRolloutOutput, A as ReleaseRowRef, At as GATE_CHECK_IDS, B as readOpencodeSessionMessages, Bt as RealnessLabels, C as scrubRolloutLine, Ct as HarborToolCall, D as FormatGateCounts, Dt as toHarborTrajectories, E as FORMAT_GATE_DISPOSITION, Et as relabelImportedSplit, F as DEFAULT_OPENCODE_DB, Ft as GateCheckedOutcome, G as claudeProjectSlug, Gt as VerifiersRolloutOutput, H as ClaudeTranscriptRef, Ht as RftItem, I as OpencodeSessionRow, It as GateEntryPoint, J as MintRolloutOptions, Jt as toJsonl, K as findClaudeTranscripts, Kt as VerifiersTokenUsage, L as findOpencodeSessionById, Lt as GatePolicy, M as gatedRolloutIds, Mt as GateCheck, N as measureFormatGate, Nt as GateCheckDisposition, O as GateDisposition, Ot as toHarborTrajectory, P as releaseRowRefs, Pt as GateCheckId, Q as ScoreOrigin, Qt as toSftRows, R as findOpencodeSessionsByDirectory, Rt as gateErrors, S as scrubLines, St as HarborSubagentTrajectoryRef, T as EmittedEvidence, Tt as fromHarborTrajectory, U as ClaudeUsageTotals, Ut as SftExportOptions, V as ClaudeTranscript, Vt as RewardRow, W as DEFAULT_CLAUDE_PROJECTS_DIR, Wt as SftRow, X as RolloutScrubber, Xt as toRftItem, Y as MintRolloutResult, Yt as toRewardRows, Z as mintRolloutRows, Zt as toRftItems, _ as ScrubCounts, _t as HarborMetrics, a as ScrubReport, at as trainingScore, b as defaultRolloutScrubber, bt as HarborStep, c as planPushCommand, ct as readRolloutLedger, d as DatasetCardInputs, dt as FromHarborOptions, en as toVerifiersRolloutOutputs, et as isRealnessGated, f as FORMAT_FILES, ft as HARBOR_IMPORT_GAP, g as SCRUB_RULES, gt as HarborImageSource, h as buildDatasetCard, ht as HarborFinalMetrics, i as RolloutReleaseCliArgs, it as trainingReward, j as assertGateReport, jt as GATE_POLICIES, k as GateReport, kt as GATE_CHECKS, l as pushDataset, lt as writeRolloutLedger, m as ReleaseFormat, mt as HarborContentPart, n as BuildSummary, nt as observedSplitScore, o as buildHfDataset, ot as appendRolloutLines, p as RELEASE_FORMATS, pt as HarborAgent, q as readClaudeTranscript, qt as realnessLabels, r as ROLLOUT_RELEASE_USAGE, rt as scoreOrigin, s as parseRolloutReleaseArgs, st as readRolloutJournal, t as BuildOptions, tt as observedScore, u as runRolloutReleaseCli, ut as ATIF_SCHEMA_VERSION, v as ScrubRule, vt as HarborObservation, w as scrubText, wt as HarborTrajectory, x as emptyScrubCounts, xt as HarborStepSource, y as addScrubCounts, yt as HarborObservationResult, z as openOpencodeDb, zt as gatedEvidenceOf } from "../index-2JJSA6-r2.js";
|
|
3
|
+
export { ATIF_SCHEMA_VERSION, type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, type EmittedEvidence, FORMAT_FILES, FORMAT_GATE_DISPOSITION, type FormatGateCounts, type FromHarborOptions, GATE_CHECKS, GATE_CHECK_IDS, GATE_POLICIES, type GateCheck, type GateCheckDisposition, type GateCheckId, type GateCheckedOutcome, type GateDisposition, type GateEntryPoint, type GatePolicy, type GateReport, type GatedEvidence, HARBOR_IMPORT_GAP, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type RealnessLabels, type ReleaseFormat, type ReleaseRowRef, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScoreOrigin, type ScorePreference, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertGateReport, assertMinted, assertMintedLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, fromHarborTrajectory, gateErrors, gateGamedOutcome, gatedEvidenceOf, gatedRolloutIds, isRealnessGated, isRolloutLine, isTrainableSplit, measureFormatGate, mintRolloutRows, observedScore, observedSplitScore, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutJournal, readRolloutLedger, realnessLabels, relabelImportedSplit, releaseRowRefs, runRolloutReleaseCli, scoreOrigin, scrubLines, scrubRolloutLine, scrubText, toHarborTrajectories, toHarborTrajectory, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, trainingReward, trainingScore, validateRolloutLine, writeRolloutLedger };
|