@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -1,2087 +1,3 @@
1
- import { DatabaseSync } from 'node:sqlite';
2
-
3
- /**
4
- * `tangle.rollout.v1` — THE canonical rollout serialization, owned by
5
- * agent-eval. One JSONL line per agent invocation (a solo eval run, a
6
- * supervisor episode, a worker session, a proposer shot, a judge call, an
7
- * analyst pass), labeled with its task/split coordinates and a single
8
- * scalar reward, carrying the FULL message transcript inline.
9
- *
10
- * This schema is the reconciliation of two prior producers:
11
- * - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
12
- * provenance hashes, the realness gate travelling into the reward,
13
- * trace-derived steps.
14
- * - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
15
- * role, task.split/rep, parent_rollout_id, policy provenance, capture
16
- * provenance, inline canonical chat-with-tools messages.
17
- * Where the two conflicted, RunRecord-derived semantics won; the wire
18
- * field names follow the ledger (snake_case). See `docs/rollout.md` for
19
- * the field-by-field decision table.
20
- *
21
- * Messages are inlined — never referenced — because every harness store a
22
- * rollout can be recovered from is mutable or garbage-collected. A line
23
- * must stay a complete training/eval example on its own.
24
- *
25
- * `outcome.reward` is THE single scalar (null = no verdict exists — a
26
- * labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
27
- * flag: a gated line must never export as a positive training example.
28
- *
29
- * That last sentence is enforced here, by `validateRolloutLine`, not merely
30
- * documented. Validating `reward` and `realness_gated` independently — each a
31
- * well-typed field, their COMBINATION unchecked — is what let a line claiming
32
- * `{reward: 0.95, realness_gated: true}` validate clean and walk into every
33
- * training export. The relationship between the two IS the invariant, so it is
34
- * checked where every other structural claim about a line is checked.
35
- *
36
- * The invariant is about the OUTCOME, not about one field of it. Zeroing
37
- * `reward` while `outcome.metrics` still carried the per-layer scores that
38
- * reward was computed from exported the gamed signal anyway, in the dict the
39
- * verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
40
- * transforms the whole outcome once, at `assertMinted` — the funnel every
41
- * minted line passes — and the reward-bearing components are relocated to
42
- * `provenance.gated_evidence`, which no exporter projects.
43
- *
44
- * WHICH checks each door applies is not decided in this file. `./gate-checks`
45
- * owns the canonical list and the total per-entry-point policy; the three doors
46
- * below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
47
- * `gateErrors` with their declared policy, so a check added to that list applies
48
- * here without anyone editing this file, and a check deliberately skipped has to
49
- * name itself there.
50
- */
51
- declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
52
- /** `agent` = a solo evaluation run (no multi-agent topology). */
53
- type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst';
54
- declare const ROLLOUT_ROLES: readonly RolloutRole[];
55
- /** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
56
- type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
57
- declare const ROLLOUT_SPLITS: readonly RolloutSplit[];
58
- /** Splits that may ship in training exports. Everything else is fail-closed excluded. */
59
- declare const TRAINABLE_SPLITS: readonly RolloutSplit[];
60
- declare function isTrainableSplit(split: RolloutSplit): boolean;
61
- /** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
62
- type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
63
- declare const ROLLOUT_CAPTURES: readonly RolloutCapture[];
64
- type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
65
- declare const CHAT_ROLES: readonly ChatRole[];
66
- interface ChatToolCall {
67
- id: string;
68
- type: 'function';
69
- function: {
70
- name: string;
71
- /** JSON-encoded argument object, exactly as the model emitted it. */
72
- arguments: string;
73
- };
74
- }
75
- interface ChatMessage {
76
- role: ChatRole;
77
- content: string | null;
78
- /** Reasoning/thinking channel where the harness captured it (full fidelity). */
79
- reasoning_content?: string;
80
- tool_calls?: ChatToolCall[];
81
- /** Required on role:"tool" — the ChatToolCall this result answers. */
82
- tool_call_id?: string;
83
- name?: string;
84
- /**
85
- * Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
86
- * from another trajectory's context, not produced by the agent on this line.
87
- * The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
88
- * on it teaches the model to author text it never authored, and credits this
89
- * run for another one's work. Absent = false (authored here).
90
- */
91
- is_copied_context?: boolean;
92
- }
93
- interface ToolDef {
94
- type: 'function';
95
- function: {
96
- name: string;
97
- description?: string;
98
- parameters?: Record<string, unknown>;
99
- };
100
- }
101
- /**
102
- * Compact trace-span projection (llm/tool step) carried alongside the
103
- * conversation when the line was minted from a trace. Optional: lines
104
- * recovered from harness stores have no span structure.
105
- */
106
- interface RolloutStep {
107
- kind: string;
108
- name: string;
109
- /** llm: last-message summary · tool: stringified args. Scrubbed. */
110
- input?: string;
111
- /** llm: output text · tool: stringified result. Scrubbed. */
112
- output?: string;
113
- status?: 'ok' | 'error';
114
- durationMs?: number;
115
- /**
116
- * LLM inferences this span represents. 0 = deterministic dispatch with no
117
- * model call — distinct from absent, which means the producer did not track it.
118
- */
119
- llm_call_count?: number;
120
- /** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
121
- prompt_token_ids?: number[];
122
- /** Exact completion tokenization; aligns index-wise with `logprobs`. */
123
- completion_token_ids?: number[];
124
- /**
125
- * Per-completion-token log probabilities under the sampling policy. Required
126
- * for off-policy correction (importance weighting) when the rollout was
127
- * generated by a policy other than the one being trained.
128
- */
129
- logprobs?: number[];
130
- }
131
- interface RolloutTask {
132
- /** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
133
- suite: string;
134
- instance_id: string;
135
- split: RolloutSplit;
136
- /** Sampling seed the campaign pinned; null = not recorded. */
137
- seed: number | null;
138
- /** Replicate index (0-based). */
139
- rep: number;
140
- }
141
- interface RolloutPolicy {
142
- /** Harness that drove the invocation (e.g. "opencode", "claude", "pi-loops"). */
143
- harness: string | null;
144
- harness_version: string | null;
145
- model: string | null;
146
- provider: string | null;
147
- /** Commit of the agent profile / candidate under evaluation. */
148
- profile_commit: string | null;
149
- /** sha256 of the effective prompt (post-steering), when recorded. */
150
- prompt_hash?: string | null;
151
- /** sha256 of the effective run config, when recorded. */
152
- config_hash?: string | null;
153
- /** Canonical agent-profile cell identity, when the run carries one. */
154
- agent_profile_cell_id?: string | null;
155
- /** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */
156
- sampling: Record<string, unknown> | null;
157
- }
158
- interface RolloutOutcome {
159
- /**
160
- * THE single scalar training signal — the official verdict.
161
- * null = no verdict exists for this invocation (a labeled gap, never 0).
162
- */
163
- reward: number | null;
164
- /** Where the reward came from (judge id; "/inherited" = parent episode's). */
165
- reward_source: string | null;
166
- /** Raw judge verdict record, verbatim. */
167
- verdict: unknown;
168
- /** Everything that is NOT the scalar reward. */
169
- metrics: Record<string, unknown>;
170
- is_completed: boolean;
171
- is_truncated: boolean;
172
- error: string | null;
173
- /**
174
- * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
175
- * its success signal. `true` requires `reward` to be 0 or null — the
176
- * validator rejects the line otherwise — and the line never qualifies for
177
- * SFT. Required on the wire: a line that does not state the flag does not
178
- * validate, so no producer can dodge the gate by omitting it.
179
- *
180
- * `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
181
- * numbers the reward was computed from are relocated to
182
- * `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
183
- * why zeroing the scalar alone was not enough.
184
- */
185
- realness_gated: boolean;
186
- /**
187
- * Whether an authenticity SCREEN ever RAN on this reward — a different claim
188
- * from `realness_gated`, which is the screen's VERDICT.
189
- *
190
- * `realness_gated: false` reads as "we looked and nothing fired". A producer
191
- * with no screen at all was emitting exactly that, so a never-screened reward
192
- * was indistinguishable on the wire from a screened-clean one, and the whole
193
- * anti-Goodhart apparatus silently treated the first as the second. The two
194
- * claims are now separable:
195
- *
196
- * - `true` — a screen ran; `realness_gated` is its verdict.
197
- * - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
198
- * `assertMinted` REFUSES such a line when its reward is above
199
- * zero: an unscreened positive reward is precisely the signal
200
- * the gate exists to qualify, and nothing has qualified it.
201
- * - absent — not stated. Pre-unification ledgers land here, as does a
202
- * `RunRecord` carrying no `outcome.realness` at all. Absent is
203
- * read as "unknown", never as `false` (which would refuse most
204
- * of the existing corpus) and never as `true`.
205
- */
206
- realness_screened?: boolean;
207
- }
208
- interface RolloutCostBlock {
209
- usd: number | null;
210
- tokens_in: number | null;
211
- tokens_out: number | null;
212
- tokens_reasoning: number | null;
213
- cache_read: number | null;
214
- cache_write: number | null;
215
- wall_s: number | null;
216
- /**
217
- * Total LLM inferences across the invocation (ATIF `llm_call_count`,
218
- * aggregated). Optional and additive: absent = not tracked, never 0.
219
- */
220
- llm_call_count?: number | null;
221
- }
222
- interface RolloutArtifacts {
223
- patch_path: string | null;
224
- run_dir: string | null;
225
- /** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
226
- transcript_ref: string | null;
227
- }
228
- /**
229
- * The reward-bearing half of a GATED line's outcome, moved off `outcome` and
230
- * parked here verbatim. Diagnostics, never training input — see
231
- * `gateGamedOutcome`.
232
- */
233
- interface GatedEvidence {
234
- /** `outcome.metrics` exactly as the producer measured it. */
235
- metrics?: Record<string, unknown>;
236
- /** `outcome.verdict` verbatim — the judge record that claimed the success. */
237
- verdict?: unknown;
238
- /**
239
- * The per-step fields `tangle.rollout.v1` does not declare, parked here when
240
- * the gate projected `steps[]` down to the schema's own key set.
241
- *
242
- * A per-step reward is training signal exactly like the scalar, and `steps`
243
- * rides through `toRewardRows` verbatim — so a gated line was shipping its
244
- * step-level credit assignment at full value beside a `reward` of 0.
245
- */
246
- steps?: unknown;
247
- }
248
- interface RolloutProvenance {
249
- captured_at: string;
250
- capture: RolloutCapture;
251
- /**
252
- * Why this line is incomplete. Required when `messages` is empty (the
253
- * transcript could not be recovered); also set by interchange importers to
254
- * name a MISSING LABEL — an imported trajectory carries no verdict, so
255
- * `outcome.reward` is null and this says why.
256
- */
257
- gap?: string;
258
- /**
259
- * Present only on a realness-gated line: the outcome fields the gate
260
- * relocated, kept so an auditor can still see WHY the run was gated and what
261
- * it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
262
- * reads `outcome` and none reads `provenance`.
263
- */
264
- gated_evidence?: GatedEvidence;
265
- }
266
- interface RolloutLine {
267
- schema: typeof ROLLOUT_SCHEMA;
268
- rollout_id: string;
269
- /** Spawning invocation within the same episode (worker → supervisor). */
270
- parent_rollout_id: string | null;
271
- run_id: string;
272
- /** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */
273
- experiment_id: string | null;
274
- /** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */
275
- candidate_id: string | null;
276
- /** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */
277
- generation: number | null;
278
- /** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */
279
- candidate_index: number | null;
280
- role: RolloutRole;
281
- task: RolloutTask;
282
- policy: RolloutPolicy;
283
- /** Full transcript, inline. [] = gap line (see provenance.gap). */
284
- messages: ChatMessage[];
285
- tool_defs: ToolDef[];
286
- /** Trace-span projections, when minted from a trace. */
287
- steps?: RolloutStep[];
288
- outcome: RolloutOutcome;
289
- cost: RolloutCostBlock;
290
- artifacts: RolloutArtifacts;
291
- provenance: RolloutProvenance;
292
- }
293
- /**
294
- * THE anti-Goodhart gate, applied to the WHOLE outcome as a TRANSFORMATION.
295
- *
296
- * Two prior rounds enforced the gate as a CHECK ON ONE FIELD at N call sites,
297
- * and each round the next reward-bearing field leaked. The one that shipped:
298
- * `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`
299
- * with no gate, so a gated run exported `reward: 0` (correct) while the
300
- * deterministic per-layer scores that reward was COMPUTED FROM — the
301
- * `layer.*` keys `rl/verifiable-reward.ts` calls the RL training signal —
302
- * shipped at 1.0, in the top-level `metrics` dict of the Prime Intellect
303
- * verifiers format, which IS that format's per-rubric score dict. `verdict`
304
- * leaks the same way into `toRftItem`'s `reference.verdict`, where a grader
305
- * author reads `resolved: true` off a run that faked it.
306
- *
307
- * So the rule is no longer "zero the field we remembered". It is: if the gate
308
- * fired, the outcome that leaves here carries NOTHING positive that was derived
309
- * from the reward, whichever field a present or future exporter decides to
310
- * read. `reward` is already forced to 0 upstream (`trainingReward`) and
311
- * REJECTED here if it is not; `metrics` and `verdict` are relocated.
312
- *
313
- * WHERE they go, and why relocation rather than deletion: zeroing destroys the
314
- * audit trail that shows why the run was gated and what it claimed, which is
315
- * the row an auditor most wants and the labeled example a gaming DETECTOR
316
- * trains on. `provenance.gated_evidence` keeps every byte, at a path no
317
- * training exporter reads — all four release configs and every `rl/exporters`
318
- * shape project from `outcome`, `messages`, `cost` and `task`; none projects
319
- * `provenance`. Auditability preserved, training signal removed, and a future
320
- * exporter that reads a field nobody thought of is safe by construction because
321
- * the field is empty rather than because the exporter remembered to check.
322
- *
323
- * Idempotent: a second application finds nothing left to move and returns the
324
- * line unchanged, so re-minting a line read back off a ledger cannot clobber
325
- * the evidence it already carries.
326
- */
327
- declare function gateGamedOutcome(line: RolloutLine): RolloutLine;
328
- declare function validateRolloutLine(value: unknown): string[];
329
- declare function assertRolloutLine(value: unknown, context?: string): asserts value is RolloutLine;
330
- declare function isRolloutLine(value: unknown): value is RolloutLine;
331
- /**
332
- * Phantom property. `declare const` means it exists only in the type system:
333
- * nothing is written at runtime, so a branded line still serializes to exactly
334
- * the same JSON as a plain one.
335
- */
336
- declare const MINTED_ROLLOUT: unique symbol;
337
- /**
338
- * A minted outcome states the gate verdict — it is not allowed to stay silent —
339
- * and, when that verdict is `true`, carries nothing else the reward was derived
340
- * from (`gateGamedOutcome` has run).
341
- */
342
- interface MintedRolloutOutcome extends RolloutOutcome {
343
- realness_gated: boolean;
344
- }
345
- /**
346
- * A `RolloutLine` whose reward has been checked against the anti-Goodhart
347
- * invariant. The type every training-data exporter takes.
348
- *
349
- * Why a brand and not just the interface: `RolloutLine` is structural, so any
350
- * hand-built object literal of the right shape IS one — which is how a line
351
- * declaring `{reward: 0.95, realness_gated: true}` reached the exporters
352
- * despite them "only accepting a minted line". The phantom symbol makes the
353
- * type nominal: it cannot be produced by writing an object literal, only by
354
- * `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
355
- * validates every line off disk), or an explicit, greppable `assertMinted`.
356
- *
357
- * Belt and braces on purpose. The brand closes first-party call sites at
358
- * COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
359
- * files, foreign imports, JSON from another process) where types are absent.
360
- * Neither alone is enough.
361
- *
362
- * Assignable to `RolloutLine` in one direction only: readers, analysis, and
363
- * the ledger writer keep taking the plain type.
364
- */
365
- type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
366
- readonly [MINTED_ROLLOUT]: true;
367
- outcome: MintedRolloutOutcome;
368
- };
369
- /**
370
- * Promote a line to the type the training exporters accept, applying the
371
- * anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch
372
- * — grep `assertMinted` to enumerate every place a line enters the training
373
- * path without coming from mint or a ledger.
374
- *
375
- * The gate runs HERE, once, rather than at each producer, because this is the
376
- * single funnel every minted line passes: `mintRolloutRows` calls it,
377
- * `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the
378
- * way out of a release, and a hand-built line has no other door. One
379
- * transformation at the funnel means an already-published ledger holding a
380
- * gated line with populated `metrics` is RE-GATED when it is read, instead of
381
- * being rejected (which would make every such artifact unreadable) or trusted
382
- * (which is the leak). Three steps, in this order:
383
- *
384
- * 1. VALIDATE the schema.
385
- * 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today
386
- * the reward relationship (which stays a REJECTION: a caller claiming
387
- * `{reward: 0.95, realness_gated: true}` is a producer defect and must fail
388
- * loudly, since laundering it into `reward: 0` here would hide the
389
- * producer) and a positive reward the producer declared it never screened.
390
- * 3. TRANSFORM the one check that policy marks `repair` — relocate the
391
- * reward's components off `outcome` (`gateGamedOutcome`), so no exporter
392
- * can leak them whichever field it reads.
393
- *
394
- * Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced
395
- * here the moment its disposition in that policy says so.
396
- *
397
- * Also normalizes the optional wire flag to an explicit boolean.
398
- * `realness_gated` is absent on pre-unification ledgers and absent means "not
399
- * flagged" per the schema, so filling it in states a claim the line was already
400
- * making, and makes the flag readable on every published row instead of most of
401
- * them. `realness_screened` is NOT filled in: absent means "unknown", and
402
- * inventing either value there would be the same overclaim this round removed.
403
- */
404
- declare function assertMinted(value: unknown, context?: string): MintedRolloutLine;
405
- /** `assertMinted` over a batch, naming the offending index in the error. */
406
- declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
407
-
408
- /**
409
- * Pure exporters over `tangle.rollout.v1` lines → the training-data shapes
410
- * the improvement loops feed:
411
- * - SFT chat JSONL (clean trainable successes, {messages, metadata})
412
- * - reward rows (every scored line, success or failure, with steps)
413
- * - Prime Intellect verifiers RolloutOutput (prompt/completion split + reward)
414
- * - OpenAI RFT items (prompt turns + verdict reference fields)
415
- *
416
- * All exporters are pure functions of the lines — filtering (never train on
417
- * holdout, reward thresholds, the realness gate) happens HERE, on inline
418
- * labels, no joins.
419
- *
420
- * Every exporter takes `MintedRolloutLine[]`, not `RolloutLine[]`: the reward
421
- * on a minted line has been checked against the anti-Goodhart invariant, and
422
- * the brand is what stops a hand-built object literal claiming a positive
423
- * reward on a gamed run from being handed to an exporter that copies it
424
- * verbatim into training data.
425
- */
426
-
427
- /**
428
- * The gate's two claims, which travel TOGETHER on every emitted row.
429
- *
430
- * `realness_gated` alone is ambiguous, and the ambiguity is exploitable:
431
- * `false` reads as "we screened it and nothing fired", so a producer that has no
432
- * screen at all emitted rows indistinguishable from screened-clean ones, and
433
- * every consumer of the published dataset read them as clean. The second field
434
- * is what separates the two claims, and it only removes the ambiguity if it
435
- * reaches the WIRE — for a round it existed on `RolloutOutcome` and on no
436
- * exported row shape at all, which left the published rows exactly as ambiguous
437
- * as before.
438
- *
439
- * So there is one helper and every row shape spreads it. A row that states one
440
- * claim without the other is not constructible by copying the pattern, and
441
- * `exporters.test.ts` walks every emitted shape to prove none does.
442
- */
443
- interface RealnessLabels {
444
- /** The screen's VERDICT: the run faked its success signal. */
445
- realness_gated: boolean;
446
- /**
447
- * Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its
448
- * verdict. `false` = the producer declares it has none. `null` = not stated
449
- * (pre-unification producers), which is "unknown" and never "clean".
450
- */
451
- realness_screened: boolean | null;
452
- }
453
- declare function realnessLabels(line: MintedRolloutLine): RealnessLabels;
454
- interface TrainingExportOptions {
455
- /** Include held-out evaluation data in training output. Default false. */
456
- allowHeldOutTrainingData?: boolean;
457
- /** Require reward to be strictly greater than this value. Default 0. */
458
- minimumQualityExclusive?: number;
459
- }
460
- /**
461
- * What a signed-signal exporter (verifiers, RFT) does with lines that are not
462
- * clean trainable successes — realness-gated lines above all.
463
- *
464
- * - 'exclude' — the default, the same fail-closed policy as every
465
- * other training export: positive, completed,
466
- * non-gated rows on a trainable split.
467
- * - 'zero-and-flag' — keep them, at their non-positive (or null) reward,
468
- * with `RealnessLabels` on the row. The dataset release
469
- * sets this per `FORMAT_GATE_DISPOSITION`: in these
470
- * formats the reward is a signed learning signal, so a
471
- * gamed trajectory at reward 0 is a correct negative,
472
- * and dropping it would bias the negative population
473
- * toward honest failures and leave a trainer no example
474
- * of gaming being penalized. The split policy is NOT
475
- * relaxed: held-out lines still need the named opt-in.
476
- *
477
- * SFT deliberately has no such option — an SFT row is an imitation target and
478
- * a gamed trajectory must never appear in one at any weight.
479
- */
480
- type GatedLineDisposition = 'exclude' | 'zero-and-flag';
481
- interface SignedSignalExportOptions extends TrainingExportOptions {
482
- /** Disposition for non-trainable lines. Default 'exclude'. */
483
- gatedLines?: GatedLineDisposition;
484
- }
485
- type SftExportOptions = TrainingExportOptions;
486
- interface SftRow {
487
- messages: ChatMessage[];
488
- metadata: {
489
- rollout_id: string;
490
- run_id: string;
491
- candidate_id: string | null;
492
- instance_id: string;
493
- reward: number;
494
- } & RealnessLabels;
495
- }
496
- /**
497
- * Supervised fine-tune rows: the completed conversation of each qualifying
498
- * line. Fail-closed filters: trainable split only (never holdout/canary),
499
- * reward strictly above `minimumQualityExclusive` (default 0), realness-gated
500
- * lines never qualify, gap lines carry no trainable content, and
501
- * copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001
502
- * rule 7 — see `ChatMessage.is_copied_context`).
503
- *
504
- * `realness_gated` is therefore always `false` on an emitted row. It is carried
505
- * anyway: an SFT row is a pure imitation target, so the row states its realness
506
- * claims instead of making the reader know the format's policy, and carrying
507
- * both flags on all four shapes is what lets the release accounting measure
508
- * every config with one rule rather than skipping the one whose row shape
509
- * happened to omit the field.
510
- */
511
- declare function toSftRows(lines: MintedRolloutLine[], options?: SftExportOptions): SftRow[];
512
- interface RewardRow {
513
- /** First user turn — the task prompt. */
514
- prompt: string;
515
- steps: RolloutStep[];
516
- reward: number;
517
- metadata: {
518
- rollout_id: string;
519
- run_id: string;
520
- candidate_id: string | null;
521
- instance_id: string;
522
- split: RolloutSplit;
523
- } & RealnessLabels;
524
- }
525
- /**
526
- * Reward-labeled rows for completed, positive-quality training runs.
527
- */
528
- declare function toRewardRows(lines: MintedRolloutLine[], options?: TrainingExportOptions): RewardRow[];
529
- interface VerifiersTokenUsage {
530
- input_tokens: number | null;
531
- output_tokens: number | null;
532
- reasoning_tokens: number | null;
533
- cache_read_tokens: number | null;
534
- cache_write_tokens: number | null;
535
- }
536
- interface VerifiersRolloutOutput {
537
- /** Messages through the last turn BEFORE the first assistant turn. */
538
- prompt: ChatMessage[];
539
- /** The first assistant turn onward — what the policy produced. */
540
- completion: ChatMessage[];
541
- reward: number | null;
542
- metrics: Record<string, unknown>;
543
- tool_defs: ToolDef[];
544
- token_usage: VerifiersTokenUsage;
545
- info: {
546
- task: RolloutLine['task'];
547
- policy: RolloutLine['policy'];
548
- rollout_id: string;
549
- run_id: string;
550
- experiment_id: string | null;
551
- candidate_id: string | null;
552
- generation: number | null;
553
- candidate_index: number | null;
554
- role: RolloutLine['role'];
555
- } & RealnessLabels;
556
- }
557
- declare function toVerifiersRolloutOutput(line: MintedRolloutLine): VerifiersRolloutOutput;
558
- declare function toVerifiersRolloutOutputs(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): VerifiersRolloutOutput[];
559
- interface RftItem {
560
- /** Prompt turns only — the graded completion is re-sampled during RFT. */
561
- messages: ChatMessage[];
562
- /** Verdict/label fields the grader references as item.reference.* */
563
- reference: {
564
- reward: number | null;
565
- reward_source: string | null;
566
- verdict: unknown;
567
- instance_id: string;
568
- suite: string;
569
- split: RolloutSplit;
570
- rollout_id: string;
571
- } & RealnessLabels;
572
- }
573
- declare function toRftItem(line: MintedRolloutLine): RftItem;
574
- /** RFT needs a real prompt: lines whose transcript starts with prompt turns. */
575
- declare function toRftItems(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): RftItem[];
576
- declare function toJsonl(rows: ReadonlyArray<unknown>): string;
577
-
578
- /**
579
- * THE canonical list of anti-Goodhart gate checks, plus the TOTAL policy every
580
- * entry point has to declare over it.
581
- *
582
- * Four rounds of adversarial review found the same defect four times, and it
583
- * was never the check itself: it was the COMPOSITION. `validateRolloutLine`
584
- * composed one check, `assertMinted` composed two, `assertRewardGate` composed
585
- * two of the three, `assertGateReport` composed its own pair — each by hand, in
586
- * its own file. So a check added to the package applied wherever its author
587
- * happened to remember, and the guard that forgot it looked exactly like the
588
- * guard that didn't. The last leak was literally that: `assertRewardGate`
589
- * composed `reward-relationship` + `gated-evidence` and not `unscreened-reward`,
590
- * so a never-screened positive reward that `assertMinted` correctly REFUSED
591
- * walked through all four waist exporters at full value.
592
- *
593
- * The fix is to make hand-composition impossible rather than to add a third
594
- * call to the two places that had two:
595
- *
596
- * - `GATE_CHECKS` is a TOTAL map over `GateCheckId`. A new id with no
597
- * implementation does not compile.
598
- * - `GatePolicy` is a TOTAL map over `GateCheckId`. Every entry point
599
- * declares one, so a new id makes EVERY entry point's policy a type error
600
- * until it is wired. Wiring it means writing `enforced`, `repairedBy(...)`
601
- * or `omittedBecause(...)` — and the last two force a written reason, so a
602
- * silent gap is not expressible.
603
- * - every check carries a `tripwire`: the minimal outcome it must refuse.
604
- * `gate-checks.test.ts` feeds each tripwire to every entry point that
605
- * declares `enforced` and requires a rejection, so wiring a check to the
606
- * wrong disposition is a TEST failure even when it type-checks.
607
- *
608
- * Adding a check is therefore: append the id, write the check, and the compiler
609
- * enumerates every place that has to decide about it.
610
- */
611
-
612
- /**
613
- * Every gate check in the package, in the order they are applied.
614
- *
615
- * Order is load-bearing only for which message a caller sees first: a line that
616
- * trips two checks reports the earlier one, and `reward-relationship` is first
617
- * because it is the invariant the other three protect.
618
- */
619
- declare const GATE_CHECK_IDS: readonly ["reward-relationship", "gated-evidence", "undeclared-step-payload", "unscreened-reward"];
620
- type GateCheckId = (typeof GATE_CHECK_IDS)[number];
621
- /**
622
- * An outcome as it reaches a check.
623
- *
624
- * Deliberately accepts a raw record as well as the typed shape: the checks are
625
- * the RUNTIME half of the gate, and the callers they exist for — JSON off a
626
- * ledger, a plain-JavaScript consumer of the published package — arrive with no
627
- * types at all. `Partial` because a tripwire states only the fields it trips on.
628
- */
629
- type GateCheckedOutcome = Partial<RolloutOutcome> | Readonly<Record<string, unknown>>;
630
- /**
631
- * What a gate check reads: the reward-bearing surface of ONE LINE.
632
- *
633
- * For three rounds the subject was the OUTCOME alone, and that assumption is
634
- * what produced the next leak rather than any missing check: `steps[]` sits on
635
- * the LINE, outside `outcome`, so a per-step reward on a gated line was read by
636
- * no check at all while `toRewardRows` copied it out verbatim — through the
637
- * MINTED door, not merely the raw one. Widening the subject is what makes
638
- * "somewhere else on the line" a place the checks can see.
639
- *
640
- * `outcome` is REQUIRED, and that is the point: a bare `RolloutOutcome` is then
641
- * not assignable to a subject, so every call site that used to pass one is a
642
- * COMPILE error until it passes the line instead. A subject with an optional
643
- * `outcome` would have let the old call sites keep compiling while silently
644
- * checking nothing — the exact failure this module exists to make impossible.
645
- */
646
- interface GateSubject {
647
- outcome: GateCheckedOutcome;
648
- /** The line's trajectory steps, when it carries any. */
649
- steps?: unknown;
650
- }
651
- interface GateCheck {
652
- id: GateCheckId;
653
- /** One sentence: what this check refuses. */
654
- refuses: string;
655
- /** One dotted-path message per defect; `[]` when the line is clean. */
656
- errors: (subject: GateSubject) => string[];
657
- /**
658
- * Every minimal subject that MUST trip `errors` — the executable form of
659
- * `refuses`, and the reason a check cannot be added without being provable.
660
- * The calibration test feeds each one to every entry point declaring
661
- * `enforced`.
662
- *
663
- * A LIST rather than one case: a check that refuses two distinct populations
664
- * (a positive reward AND a reward it cannot read as a number) proved able to
665
- * hold for the first while silently passing the second, so each population
666
- * states its own tripwire and each is exercised separately.
667
- */
668
- tripwires: GateSubject[];
669
- }
670
- /**
671
- * The reward-bearing outcome fields that are NOT the scalar: the numbers the
672
- * reward was computed from, and the verdict record that claimed it.
673
- *
674
- * Returned as one block rather than filtered key-by-key. A key-name heuristic
675
- * ("zero anything matching `layer.*` or `/score/`") is the same defect shape as
676
- * the line-oriented regex the AST score guard replaced: it holds until someone
677
- * names a metric `pass_fraction`, and the next reward-shaped key ships at full
678
- * value. The producer's OWN classification — "this is the scalar, that is
679
- * everything else" — is the only partition that cannot be out-guessed.
680
- */
681
- declare function gatedEvidenceOf(subject: GateSubject): GatedEvidence | undefined;
682
- /**
683
- * The registry. Total over `GateCheckId`, so an id with no check does not
684
- * compile, and `GATE_CHECK_IDS` stays the single enumeration everything
685
- * iterates.
686
- */
687
- declare const GATE_CHECKS: {
688
- readonly [K in GateCheckId]: GateCheck;
689
- };
690
- /**
691
- * What ONE entry point does about ONE check.
692
- *
693
- * `repair` and `omit` both carry a mandatory sentence, which is the mechanism
694
- * that keeps a legitimate omission distinguishable from a forgotten one: you
695
- * cannot skip a check without writing down why, and the reasons are readable
696
- * side by side in `GATE_POLICIES`.
697
- */
698
- type GateCheckDisposition = {
699
- readonly kind: 'enforce';
700
- }
701
- /** Resolved by TRANSFORMING the line instead of rejecting it; `by` names the function. */
702
- | {
703
- readonly kind: 'repair';
704
- readonly by: string;
705
- }
706
- /** Deliberately not applied here; `because` states the reason. */
707
- | {
708
- readonly kind: 'omit';
709
- readonly because: string;
710
- };
711
- /** Total over `GateCheckId`: a new check makes every policy literal a type error. */
712
- type GatePolicy = {
713
- readonly [K in GateCheckId]: GateCheckDisposition;
714
- };
715
- /**
716
- * Every entry point that decides about the gate, and what it decides.
717
- *
718
- * Read this as the package's gate policy in one screen. The four entry points
719
- * are not interchangeable — a validator that rejects, a mint funnel that
720
- * repairs, a runtime backstop for untyped callers, and a release certifier over
721
- * emitted rows — and the dispositions say which is which.
722
- */
723
- declare const GATE_POLICIES: {
724
- /**
725
- * The schema validator. Rejects the reward relationship and NOTHING ELSE, on
726
- * purpose: it runs on every line read off disk, and the other two conditions
727
- * describe artifacts that already exist.
728
- */
729
- readonly validateRolloutLine: {
730
- readonly 'reward-relationship': {
731
- readonly kind: "enforce";
732
- };
733
- readonly 'gated-evidence': GateCheckDisposition;
734
- readonly 'undeclared-step-payload': GateCheckDisposition;
735
- readonly 'unscreened-reward': GateCheckDisposition;
736
- };
737
- /**
738
- * The mint funnel — the single door every `MintedRolloutLine` passes. Validates
739
- * first (so the reward relationship has already been rejected), then refuses
740
- * what cannot be repaired, then repairs what can.
741
- */
742
- readonly assertMinted: {
743
- readonly 'reward-relationship': {
744
- readonly kind: "enforce";
745
- };
746
- readonly 'gated-evidence': GateCheckDisposition;
747
- readonly 'undeclared-step-payload': GateCheckDisposition;
748
- readonly 'unscreened-reward': {
749
- readonly kind: "enforce";
750
- };
751
- };
752
- /**
753
- * The runtime backstop, and the one entry point with no license to omit
754
- * anything: it exists for callers the type system never saw (plain JavaScript
755
- * handing an object literal to a published exporter), so a check it skips is a
756
- * check that does not run at all for them. This is where the fourth leak was.
757
- */
758
- readonly assertRewardGate: {
759
- readonly 'reward-relationship': {
760
- readonly kind: "enforce";
761
- };
762
- readonly 'gated-evidence': {
763
- readonly kind: "enforce";
764
- };
765
- readonly 'undeclared-step-payload': {
766
- readonly kind: "enforce";
767
- };
768
- readonly 'unscreened-reward': {
769
- readonly kind: "enforce";
770
- };
771
- };
772
- /**
773
- * The release certifier. Same checks, measured over the rows a release is
774
- * ABOUT TO WRITE rather than over one line's outcome — see `REPORT_MEASURES`
775
- * in `release/gate-report.ts`, which is the second total map this policy
776
- * drives.
777
- */
778
- readonly assertGateReport: {
779
- readonly 'reward-relationship': {
780
- readonly kind: "enforce";
781
- };
782
- readonly 'gated-evidence': {
783
- readonly kind: "enforce";
784
- };
785
- readonly 'undeclared-step-payload': {
786
- readonly kind: "enforce";
787
- };
788
- readonly 'unscreened-reward': {
789
- readonly kind: "enforce";
790
- };
791
- };
792
- };
793
- /** Every entry point that declares a gate policy. */
794
- type GateEntryPoint = keyof typeof GATE_POLICIES;
795
- /**
796
- * Run the checks one entry point enforces. The ONLY way an entry point should
797
- * obtain gate errors — hand-composing two of the three is the bug this module
798
- * exists to remove.
799
- */
800
- declare function gateErrors(subject: GateSubject, policy: GatePolicy): string[];
801
-
802
- /**
803
- * Harbor ATIF-v1.7 interchange — `tangle.rollout.v1` ⇄ Agent Trajectory
804
- * Interchange Format.
805
- *
806
- * ATIF is the portability format (spec:
807
- * https://www.harborframework.com/docs/agents/trajectory-format, normative
808
- * RFC: harbor-framework/harbor `rfcs/0001-trajectory-format.md`). It sits
809
- * BELOW the waist of the rollout hourglass in both directions — export reads
810
- * `RolloutLine[]`, import writes `RolloutLine[]` — and it is never a source
811
- * of training labels:
812
- *
813
- * ATIF models NO reward, NO judge verdict, NO task/split coordinates.
814
- *
815
- * Consequences, both deliberate:
816
- * - EXPORT drops `outcome.reward`, `outcome.reward_source` and
817
- * `outcome.verdict` entirely. They are not smuggled into `extra`: a
818
- * third-party reading our ATIF file must not be able to mistake an
819
- * agent-eval judge score for something ATIF sanctioned.
820
- * - IMPORT therefore mints UNLABELED lines: `reward: null` (the existing
821
- * "null reward is a labeled gap, never 0" semantics), `verdict: null`,
822
- * and a `provenance.gap` naming the missing label. An imported
823
- * trajectory is not a training example until a judge scores it.
824
- *
825
- * Everything else we own that ATIF has no field for travels in a namespaced
826
- * escrow at `extra.tangle.*`, so our own round-trip is exact while a foreign
827
- * reader can ignore it. Fields that neither ATIF nor the escrow can carry
828
- * come back explicitly null / fail-closed, never invented.
829
- *
830
- * THE ESCROW IS NAMESPACED, NOT AUTHENTICATED. Anyone can write
831
- * `extra.tangle.*` into a file. So the escrow may restore what a value IS, but
832
- * never what a line is ALLOWED to do: `task.split` is forced to `holdout` on
833
- * every import regardless of what the document claims, and promoting an
834
- * imported trajectory to a trainable split is an explicit, greppable act
835
- * (`relabelImportedSplit`) rather than a property of the file. The document
836
- * keeps its claim — the claim just is not authority.
837
- *
838
- * Multi-agent shape differs on purpose. ATIF EMBEDS children in
839
- * `subagent_trajectories`; we keep a flat ledger with a normalized
840
- * `parent_rollout_id` edge. Export assembles the tree, import flattens it.
841
- * `session_id` is RUN-scoped in ATIF, so it carries `run_id` — the coordinate
842
- * that is shared by every invocation of one run — not `rollout_id`, which
843
- * identifies a single invocation and would split one run across session ids.
844
- *
845
- * ROUND-TRIPPING IS IDEMPOTENT: `import(export(import(export(x))))` is
846
- * byte-identical to `import(export(x))`. Import composes `provenance.gap` as a
847
- * de-duplicated ordered set rather than appending, and it emits every
848
- * `ChatMessage` with keys in the canonical schema order (role, content,
849
- * reasoning_content, tool_calls, tool_call_id, name, is_copied_context), so a
850
- * ledger hashed on serialized bytes sees no diff across further passes. The
851
- * FIRST import may re-order a producer's keys — that is the canonicalization.
852
- *
853
- * NOT building a Letta converter. Letta's trajectory-v1 is a strict subset of
854
- * what we need from ATIF here — no per-step or aggregate cost, no
855
- * multi-agent/subagent structure, no token-id or logprob channel — so a Letta
856
- * sink would carry less than this one and add a second format to keep
857
- * correct. Decision recorded in docs/rollout.md; do not re-litigate without a
858
- * concrete consumer that reads Letta and cannot read ATIF.
859
- */
860
-
861
- declare const ATIF_SCHEMA_VERSION = "ATIF-v1.7";
862
- /** Gap note on every imported line — ATIF carries no verdict, so nothing is scored. */
863
- declare const HARBOR_IMPORT_GAP = "imported from Harbor ATIF; no verdict";
864
- type HarborStepSource = 'system' | 'user' | 'agent';
865
- interface HarborImageSource {
866
- media_type: string;
867
- path: string;
868
- }
869
- interface HarborContentPart {
870
- type: 'text' | 'image';
871
- text?: string;
872
- source?: HarborImageSource;
873
- }
874
- interface HarborToolCall {
875
- tool_call_id: string;
876
- function_name: string;
877
- /** ATIF requires a decoded JSON object here, unlike our raw argument string. */
878
- arguments: Record<string, unknown>;
879
- extra?: Record<string, unknown>;
880
- }
881
- interface HarborSubagentTrajectoryRef {
882
- trajectory_id?: string;
883
- trajectory_path?: string;
884
- /** Informational only since v1.7 — never a resolution key. */
885
- session_id?: string;
886
- extra?: Record<string, unknown>;
887
- }
888
- interface HarborObservationResult {
889
- source_call_id?: string;
890
- content?: string | HarborContentPart[];
891
- subagent_trajectory_ref?: HarborSubagentTrajectoryRef[];
892
- extra?: Record<string, unknown>;
893
- }
894
- interface HarborObservation {
895
- results: HarborObservationResult[];
896
- }
897
- interface HarborMetrics {
898
- prompt_tokens?: number;
899
- completion_tokens?: number;
900
- cached_tokens?: number;
901
- cost_usd?: number;
902
- prompt_token_ids?: number[];
903
- completion_token_ids?: number[];
904
- logprobs?: number[];
905
- extra?: Record<string, unknown>;
906
- }
907
- interface HarborStep {
908
- /** Ordinal, sequential from 1. */
909
- step_id: number;
910
- timestamp?: string;
911
- source: HarborStepSource;
912
- model_name?: string;
913
- reasoning_effort?: string | number;
914
- message: string | HarborContentPart[];
915
- reasoning_content?: string;
916
- tool_calls?: HarborToolCall[];
917
- observation?: HarborObservation;
918
- metrics?: HarborMetrics;
919
- llm_call_count?: number;
920
- is_copied_context?: boolean;
921
- extra?: Record<string, unknown>;
922
- }
923
- interface HarborAgent {
924
- name: string;
925
- version: string;
926
- model_name?: string;
927
- /** OpenAI function-calling schema — byte-identical to our `ToolDef`. */
928
- tool_definitions?: ToolDef[];
929
- extra?: Record<string, unknown>;
930
- }
931
- interface HarborFinalMetrics {
932
- total_prompt_tokens?: number;
933
- total_completion_tokens?: number;
934
- total_cached_tokens?: number;
935
- total_cost_usd?: number;
936
- total_steps?: number;
937
- extra?: Record<string, unknown>;
938
- }
939
- interface HarborTrajectory {
940
- schema_version: string;
941
- session_id?: string;
942
- /** Required on embedded subagents; we always set it so lines stay joinable. */
943
- trajectory_id?: string;
944
- agent: HarborAgent;
945
- steps: HarborStep[];
946
- notes?: string;
947
- final_metrics?: HarborFinalMetrics;
948
- continued_trajectory_ref?: string;
949
- subagent_trajectories?: HarborTrajectory[];
950
- extra?: Record<string, unknown>;
951
- }
952
- /**
953
- * Assemble one episode's flat lines into a single ATIF trajectory tree,
954
- * linked by `parent_rollout_id`.
955
- *
956
- * Reward, verdict and split are NOT emitted (ATIF models none of them); the
957
- * split and the rest of the task coordinates survive only in `extra.tangle`.
958
- *
959
- * We deliberately do NOT synthesize an `observation.subagent_trajectory_ref`
960
- * pointing at each child: our ledger records WHICH invocation spawned a
961
- * worker, not which STEP did, and attaching the ref to a guessed step would
962
- * fabricate a causal claim. Children are embedded in `subagent_trajectories`
963
- * (each with the `trajectory_id` the spec requires) and the edge is stated in
964
- * the child's escrowed `parent_rollout_id`.
965
- *
966
- * Throws when the lines are not one tree — use `toHarborTrajectories` for a forest.
967
- */
968
- declare function toHarborTrajectory(lines: RolloutLine[]): HarborTrajectory;
969
- /** Every independent tree in the input, one ATIF document each. */
970
- declare function toHarborTrajectories(lines: RolloutLine[]): HarborTrajectory[];
971
- interface FromHarborOptions {
972
- /** Injected clock for deterministic output when the source carries no capture time. */
973
- now?: () => Date;
974
- }
975
- /**
976
- * Flatten an ATIF trajectory tree back into `tangle.rollout.v1` lines, parent
977
- * first, each child carrying `parent_rollout_id`.
978
- *
979
- * Every line comes back UNLABELED: `reward`, `reward_source` and `verdict` are
980
- * null and `provenance.gap` says why. ATIF models no verdict, so scoring an
981
- * imported trajectory is a judge's job, not this function's. Every line lands
982
- * on `holdout` whatever the document claims — see `relabelImportedSplit`.
983
- */
984
- declare function fromHarborTrajectory(trajectory: HarborTrajectory, options?: FromHarborOptions): RolloutLine[];
985
- /**
986
- * THE explicit door out of `holdout` for imported lines.
987
- *
988
- * Import forces `holdout` because a document's own claim about its split is not
989
- * evidence — anyone can write `extra.tangle.task.split`. Promoting a file to a
990
- * trainable split is an operator's decision about provenance they verified, so
991
- * it is a separate, greppable call: `grep relabelImportedSplit` enumerates
992
- * every place foreign data was declared trainable, which is exactly the audit
993
- * the trusted-escrow version made impossible.
994
- *
995
- * Returns plain `RolloutLine`s. They still have to pass `assertMinted` (and its
996
- * anti-Goodhart check) to reach an exporter — re-labeling a split is not
997
- * minting a reward.
998
- */
999
- declare function relabelImportedSplit(lines: readonly RolloutLine[], split: RolloutSplit): RolloutLine[];
1000
-
1001
- /**
1002
- * Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`
1003
- * lines. Writes validate BEFORE touching disk (a bad line never lands);
1004
- * reads validate line-by-line and fail loud with the line number, because a
1005
- * silently-skipped rollout is a corrupted dataset.
1006
- *
1007
- * "Validate" includes the anti-Goodhart invariant (a realness-gated line may
1008
- * not carry a positive reward), so a poisoned line can neither enter a ledger
1009
- * nor leave one.
1010
- *
1011
- * Two read modes, matching the two write-side row classes: `readRolloutLedger`
1012
- * re-validates under the mint policy (training data), `readRolloutJournal`
1013
- * under the write policy (supervision journals, whose unscreened positive
1014
- * rewards are writable and must stay readable).
1015
- */
1016
-
1017
- /** Replace the ledger file with exactly `lines`. */
1018
- declare function writeRolloutLedger(path: string, lines: RolloutLine[]): Promise<void>;
1019
- /** Append `lines` to the ledger file (created if absent). */
1020
- declare function appendRolloutLines(path: string, lines: RolloutLine[]): Promise<void>;
1021
- /**
1022
- * Read and validate every line. Throws on the first malformed/invalid line
1023
- * (with its 1-based line number) — fail-closed, never a silent drop.
1024
- *
1025
- * Validation includes the anti-Goodhart invariant, which is why the result is
1026
- * `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this
1027
- * process from outside the type system (another run, another machine, a
1028
- * hand-edited JSONL), so this read is the runtime boundary where a poisoned
1029
- * line is refused rather than exported.
1030
- */
1031
- declare function readRolloutLedger(path: string): Promise<MintedRolloutLine[]>;
1032
- /**
1033
- * Read a ledger under the WRITE-side policy (`validateRolloutLine`), which
1034
- * omits the unscreened-reward check. `writeRolloutLedger` accepts a
1035
- * supervision-journal row (`realness_screened: false` with a positive reward
1036
- * — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says
1037
- * such rows "must stay writable, readable and reportable"; a read API that
1038
- * only re-validated under `assertMinted` made every such file unreadable —
1039
- * write-accepted but read-refused is a data-loss trap.
1040
- *
1041
- * The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here
1042
- * can reach a training exporter without passing `assertMinted`, so the
1043
- * promotion gate (which DOES enforce unscreened-reward) is exactly as closed
1044
- * as before. Use `readRolloutLedger` when the file is training data.
1045
- */
1046
- declare function readRolloutJournal(path: string): Promise<RolloutLine[]>;
1047
-
1048
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
1049
- type AgentProfileDimensionValue = string | number | boolean | null;
1050
- interface AgentProfileSource {
1051
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
1052
- kind: string;
1053
- /** sha256 over the canonical source profile object. */
1054
- hash: string;
1055
- }
1056
- interface AgentProfileHarness {
1057
- id: string;
1058
- version?: string;
1059
- hash?: string;
1060
- }
1061
- interface AgentProfileCell {
1062
- schemaVersion: AgentProfileCellSchemaVersion;
1063
- cellId: string;
1064
- profileId: string;
1065
- sourceProfile: AgentProfileSource;
1066
- harness?: AgentProfileHarness;
1067
- model?: string;
1068
- promptHash?: string;
1069
- dimensions?: Record<string, AgentProfileDimensionValue>;
1070
- }
1071
-
1072
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
1073
- interface BudgetSpec {
1074
- tokens?: number;
1075
- wallMs?: number;
1076
- calls?: number;
1077
- usd?: number;
1078
- }
1079
- interface RunOutcome$1 {
1080
- score?: number;
1081
- pass?: boolean;
1082
- failureClass?: FailureClass;
1083
- notes?: string;
1084
- }
1085
- /**
1086
- * Layer — optional classification in a nested build workflow.
1087
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
1088
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
1089
- * `app-runtime`: a run of the generated agent against a domain scenario.
1090
- * `meta`: any meta-eval (judge replay, correlation analysis).
1091
- */
1092
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
1093
- interface Run {
1094
- runId: string;
1095
- /**
1096
- * Stable identifier of the scenario being executed.
1097
- *
1098
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
1099
- * input WITHOUT this field, substituting a sensible default
1100
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
1101
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
1102
- * keeps the persisted shape unambiguous for downstream filters + aggregations
1103
- * while removing the boilerplate of inventing placeholder ids at the call site.
1104
- */
1105
- scenarioId: string;
1106
- variantId?: string;
1107
- datasetVersion?: string;
1108
- /** Git SHA of agent code at run time. */
1109
- codeSha?: string;
1110
- /** Hash of the prompt template + any system prompt. */
1111
- promptSha?: string;
1112
- /** Model id + date + system-prompt hash, concatenated. */
1113
- modelFingerprint?: string;
1114
- seed?: number;
1115
- /** Arbitrary environment markers (shell, docker version, tz). */
1116
- envFingerprint?: Record<string, string>;
1117
- /** Version of the redaction rules applied to this run. */
1118
- redactionVersion?: string;
1119
- /** Parent run in a nested build workflow. A builder run's children are
1120
- * app-build runs; those children are app-runtime runs. */
1121
- parentRunId?: string;
1122
- /** Stable project identifier — groups runs across chats + sessions. */
1123
- projectId?: string;
1124
- /** Chat/conversation identifier within a project. */
1125
- chatId?: string;
1126
- /** Layer classification — hint for aggregation; not enforced. */
1127
- layer?: RunLayer;
1128
- startedAt: number;
1129
- endedAt?: number;
1130
- status: RunStatus;
1131
- outcome?: RunOutcome$1;
1132
- budget?: BudgetSpec;
1133
- /** Free-form labels for downstream grouping. */
1134
- tags?: Record<string, string>;
1135
- }
1136
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
1137
- type SpanStatus = 'ok' | 'error';
1138
- interface SpanBase {
1139
- spanId: string;
1140
- parentSpanId?: string;
1141
- runId: string;
1142
- kind: SpanKind;
1143
- name: string;
1144
- startedAt: number;
1145
- endedAt?: number;
1146
- status?: SpanStatus;
1147
- error?: string;
1148
- /** Anything not covered by typed fields. Kept deliberately free-form. */
1149
- attributes?: Record<string, unknown>;
1150
- }
1151
- interface Message {
1152
- role: 'system' | 'user' | 'assistant' | 'tool';
1153
- content: string;
1154
- tokens?: number;
1155
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
1156
- images?: Array<{
1157
- artifactId?: string;
1158
- url?: string;
1159
- mime?: string;
1160
- }>;
1161
- }
1162
- interface LlmSpan extends SpanBase {
1163
- kind: 'llm';
1164
- model: string;
1165
- messages: Message[];
1166
- output?: string;
1167
- inputTokens?: number;
1168
- /** All generated tokens, including the reasoning subset when present. */
1169
- outputTokens?: number;
1170
- cachedTokens?: number;
1171
- cacheWriteTokens?: number;
1172
- /** Reasoning-token subset of `outputTokens`. */
1173
- reasoningTokens?: number;
1174
- costUsd?: number;
1175
- finishReason?: string;
1176
- }
1177
- interface ToolSpan extends SpanBase {
1178
- kind: 'tool';
1179
- toolName: string;
1180
- args: unknown;
1181
- /** False when the source observed the call but did not capture its arguments. */
1182
- argsCaptured?: boolean;
1183
- result?: unknown;
1184
- latencyMs?: number;
1185
- }
1186
- interface RetrievalSpan extends SpanBase {
1187
- kind: 'retrieval';
1188
- query: string;
1189
- hits: Array<{
1190
- docId: string;
1191
- score: number;
1192
- content?: string;
1193
- }>;
1194
- }
1195
- interface JudgeSpan extends SpanBase {
1196
- kind: 'judge';
1197
- judgeId: string;
1198
- /** Span this judgment applies to. */
1199
- targetSpanId: string;
1200
- dimension: string;
1201
- /** Numeric score (free-range; interpretation up to the judge). */
1202
- score: number;
1203
- rationale?: string;
1204
- evidence?: string;
1205
- }
1206
- interface SandboxSpan extends SpanBase {
1207
- kind: 'sandbox';
1208
- image?: string;
1209
- command?: string;
1210
- exitCode?: number;
1211
- testsTotal?: number;
1212
- testsPassed?: number;
1213
- stdoutHash?: string;
1214
- stderrHash?: string;
1215
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
1216
- wallMs?: number;
1217
- }
1218
- interface GenericSpan extends SpanBase {
1219
- kind: 'agent' | 'custom';
1220
- }
1221
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
1222
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
1223
- interface TraceEvent {
1224
- eventId: string;
1225
- runId: string;
1226
- spanId?: string;
1227
- kind: EventKind;
1228
- timestamp: number;
1229
- payload: Record<string, unknown>;
1230
- }
1231
- interface BudgetLedgerEntry {
1232
- runId: string;
1233
- dimension: keyof BudgetSpec;
1234
- limit: number;
1235
- consumed: number;
1236
- remaining: number;
1237
- timestamp: number;
1238
- breached: boolean;
1239
- /** Span that triggered this entry, if any. */
1240
- spanId?: string;
1241
- }
1242
- interface Artifact {
1243
- artifactId: string;
1244
- runId: string;
1245
- spanId?: string;
1246
- contentType: string;
1247
- sizeBytes: number;
1248
- /** sha256 in hex. */
1249
- hash: string;
1250
- /** External storage URL (R2, S3, filesystem path). */
1251
- storageUrl?: string;
1252
- /** Inline content for small blobs — keep under ~64KB. */
1253
- inlineContent?: string;
1254
- }
1255
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
1256
-
1257
- /**
1258
- * Paper-grade RunRecord schema + runtime validator.
1259
- *
1260
- * Every run that participates in a promotion gate, paper table, or
1261
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
1262
- * fields are exactly those the paper "Two Loops, Three Roles" requires
1263
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
1264
- * holdout split tag. A task score is optional because execution-only records
1265
- * must preserve missing labels instead of converting errors into zero quality.
1266
- *
1267
- * This is intentionally NOT a replacement for the rich `Run` /
1268
- * `ProposeReviewReport` / `ScenarioResult` types already in the
1269
- * package. Those are runtime structures with full provenance. A
1270
- * `RunRecord` is the analysis-time projection — the JSON-friendly
1271
- * row you'd put in a parquet file or paste into a notebook.
1272
- *
1273
- * Validate at the boundary:
1274
- *
1275
- * const rec = validateRunRecord(rawJson) // throws on missing
1276
- * const ok = isRunRecord(rawJson) // boolean check
1277
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
1278
- *
1279
- * The validator runs in pure TS — zod is intentionally NOT a
1280
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
1281
- */
1282
-
1283
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
1284
- * combined train+test pool that the optimizer is allowed to read. */
1285
- type RunSplitTag = 'search' | 'dev' | 'holdout';
1286
- /**
1287
- * Explicit execution-lifecycle result for a run.
1288
- *
1289
- * This is separate from task quality (`outcome`) and failure classification.
1290
- * Producers set it only from root-run or process evidence.
1291
- */
1292
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
1293
- interface RunTokenUsage {
1294
- input: number;
1295
- /** All generated tokens charged as output, including reasoning tokens. */
1296
- output: number;
1297
- /** Reasoning-token subset of `output`, when the provider reports it. */
1298
- reasoning?: number;
1299
- /** Prompt tokens served from a provider cache. */
1300
- cached?: number;
1301
- /** Prompt tokens written into a provider cache. */
1302
- cacheWrite?: number;
1303
- }
1304
- /**
1305
- * How a run's USD amount was obtained.
1306
- */
1307
- type RunCostProvenance = {
1308
- kind: 'observed';
1309
- usd: number;
1310
- } | {
1311
- kind: 'estimated';
1312
- usd: number;
1313
- } | {
1314
- kind: 'uncaptured';
1315
- usd: null;
1316
- };
1317
- interface RunJudgeMetadata {
1318
- model: string;
1319
- promptVersion: string;
1320
- /** [0,1] confidence the judge declared. Constant judge confidence
1321
- * across many runs is a fallback signal (see `canary.ts`). */
1322
- confidence: number;
1323
- /** True if the judge degraded to a fallback path (rules-only,
1324
- * prior-call cache, etc.). The canary uses this to alert. */
1325
- fallback: boolean;
1326
- }
1327
- /**
1328
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
1329
- * judges over a multi-dimensional rubric.
1330
- *
1331
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
1332
- * composite the gate uses. The full breakdown belongs here so consumers
1333
- * can answer "which judge disagreed?", "which dimension dragged the
1334
- * composite down?", and "did half the panel fail?" without re-running.
1335
- *
1336
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
1337
- * `composite` are convenience projections — derivable but precomputed so
1338
- * downstream IRR primitives (`interRaterReliability`,
1339
- * `corpusInterRaterAgreement`) and reporters don't pay the same
1340
- * aggregation twice.
1341
- *
1342
- * Fail-loud discipline: judges that errored out land in `failedJudges`
1343
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
1344
- * run); the explicit list makes a partial-failure recorded as such.
1345
- */
1346
- interface JudgeScoresRecord {
1347
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
1348
- perJudge: Record<string, Record<string, number>>;
1349
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
1350
- perDimMean: Record<string, number>;
1351
- /** Composite mean across successful judges. Mirrors the task score only
1352
- * when `failedJudges` is empty. */
1353
- composite: number;
1354
- /** Judges that errored or returned an unparseable verdict. Recorded
1355
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
1356
- * not inferred from missing keys in `perJudge`. */
1357
- failedJudges?: string[];
1358
- /** Free-form notes the judges emitted (joined across judges or
1359
- * first-judge only — consumer's choice). */
1360
- notes?: string;
1361
- }
1362
- interface RunOutcome {
1363
- /** Score on the search/optimization split. Optional for holdout-only and
1364
- * execution-only records. */
1365
- searchScore?: number;
1366
- /** Score on the held-out split. Optional for search-only and execution-only
1367
- * records. When both scores are absent, the run is explicitly unlabeled. */
1368
- holdoutScore?: number;
1369
- /** Bag of any other metric the run produced — judge dimensions,
1370
- * pass/fail counters, latency stats, etc. Numeric only — keeps
1371
- * reporters honest. */
1372
- raw: Record<string, number>;
1373
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
1374
- * judgements populate this; substrate primitives like
1375
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
1376
- * these records as input. Optional — single-judge or scalar-only
1377
- * runs leave it unset. */
1378
- judgeScores?: JudgeScoresRecord;
1379
- /** Authenticity / realness verdict — did the run build the REAL thing on the
1380
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
1381
- * with an authenticity config populate it. Carried in the corpus so the
1382
- * flywheel / off-policy learning can optimize for real completion, not gamed
1383
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
1384
- * must not count as a real success regardless of `score`. */
1385
- realness?: {
1386
- score: number;
1387
- gated: boolean;
1388
- reason?: string;
1389
- };
1390
- }
1391
- /**
1392
- * Mandatory paper-grade fields for a single evaluation run. Optional
1393
- * fields are extension points; mandatory fields throw if missing.
1394
- *
1395
- * Hash discipline:
1396
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
1397
- * model (after any steering bundle merge).
1398
- * - `configHash` is the sha256 of the effective run config (model,
1399
- * temperature, tools, judges, splits). The pair (promptHash,
1400
- * configHash) uniquely identifies an experiment cell.
1401
- *
1402
- * Model snapshot discipline:
1403
- * - `model` MUST encode a snapshot version. Bare aliases like
1404
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
1405
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
1406
- */
1407
- interface RunRecord {
1408
- /** UUID for the run. */
1409
- runId: string;
1410
- /** Logical experiment grouping (a treatment vs a baseline within
1411
- * the same sweep should share `experimentId`). */
1412
- experimentId: string;
1413
- /** Stable identifier for the candidate (variant) being run. The
1414
- * promotion gate compares two `candidateId`s on matched items. */
1415
- candidateId: string;
1416
- /** RNG seed for the run. Always recorded — silent re-seeding is
1417
- * the most common cause of non-reproducible numbers. */
1418
- seed: number;
1419
- /** Model identifier WITH snapshot version. */
1420
- model: string;
1421
- /** sha256 of the effective prompt (post-steering). */
1422
- promptHash: string;
1423
- /** sha256 of the effective config. */
1424
- configHash: string;
1425
- /** Git SHA the harness was run from. */
1426
- commitSha: string;
1427
- /** End-to-end wall-clock duration in milliseconds. */
1428
- wallMs: number;
1429
- /** Time spent queued before execution started, if known. */
1430
- queueMs?: number;
1431
- /** Total USD cost, or null when the producer could not capture one. */
1432
- costUsd: number | null;
1433
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
1434
- costProvenance: RunCostProvenance;
1435
- /** Token usage breakdown. */
1436
- tokenUsage: RunTokenUsage;
1437
- /** Root-run or process terminal result. Never inferred from a child span. */
1438
- terminalOutcome: RunTerminalOutcome;
1439
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
1440
- * or incomplete terminal result; never populated from a child span. */
1441
- terminalFailureReason?: string;
1442
- /** Judge-side metadata, if a judge was used. */
1443
- judgeMetadata?: RunJudgeMetadata;
1444
- /** Per-split scores + raw bag. */
1445
- outcome: RunOutcome;
1446
- /** Canonical task-failure class drawn from the shared
1447
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
1448
- * evidence. Execution errors belong in
1449
- * `outcome.raw.execution_error_count`. */
1450
- failureClass?: FailureClass;
1451
- /** Free-form task-failure detail scoped under a non-success
1452
- * `failureClass`. It is invalid without that class. */
1453
- failureMode?: string;
1454
- /** Which split this run was drawn from. */
1455
- splitTag: RunSplitTag;
1456
- /**
1457
- * Stable scenario identifier the run observed or was scored against.
1458
- * Comparison primitives match this identity rather than input order.
1459
- */
1460
- scenarioId: string;
1461
- /**
1462
- * Canonical identity for the agent profile cell that produced this row:
1463
- * profile artifact hash plus optional harness/model/prompt/reporting
1464
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
1465
- * longitudinal reports by the complete source profile, not by a loose
1466
- * candidate label or opaque config hash.
1467
- */
1468
- agentProfile?: AgentProfileCell;
1469
- }
1470
-
1471
- interface RunFilter {
1472
- scenarioId?: string;
1473
- variantId?: string;
1474
- status?: RunStatus;
1475
- since?: number;
1476
- until?: number;
1477
- tag?: {
1478
- key: string;
1479
- value: string;
1480
- };
1481
- parentRunId?: string;
1482
- projectId?: string;
1483
- chatId?: string;
1484
- layer?: RunLayer;
1485
- }
1486
- interface SpanFilter {
1487
- runId?: string;
1488
- parentSpanId?: string;
1489
- kind?: SpanKind;
1490
- name?: string;
1491
- toolName?: string;
1492
- judgeId?: string;
1493
- since?: number;
1494
- until?: number;
1495
- }
1496
- interface EventFilter {
1497
- runId?: string;
1498
- spanId?: string;
1499
- kind?: EventKind;
1500
- since?: number;
1501
- until?: number;
1502
- }
1503
- interface TraceStore {
1504
- appendRun(run: Run): Promise<void>;
1505
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
1506
- appendSpan(span: Span): Promise<void>;
1507
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
1508
- appendEvent(event: TraceEvent): Promise<void>;
1509
- appendArtifact(artifact: Artifact): Promise<void>;
1510
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
1511
- getRun(runId: string): Promise<Run | undefined>;
1512
- listRuns(filter?: RunFilter): Promise<Run[]>;
1513
- spans(filter?: SpanFilter): Promise<Span[]>;
1514
- events(filter?: EventFilter): Promise<TraceEvent[]>;
1515
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
1516
- artifacts(runId: string): Promise<Artifact[]>;
1517
- }
1518
-
1519
- /**
1520
- * The two named score derivations every consumer must choose between.
1521
- *
1522
- * The anti-Goodhart gate (`outcome.realness.gated`) only holds if it is
1523
- * impossible to read a run's score WITHOUT deciding whether the gate applies.
1524
- * A bare `outcome.holdoutScore ?? outcome.searchScore` makes that decision
1525
- * invisible — and silently answers "no gate", which is the wrong default on
1526
- * every path that produces training data. So the expression lives here, once,
1527
- * behind two names that force the caller to state the intent:
1528
- *
1529
- * - `trainingScore` / `trainingReward` — GATED. Anything that becomes
1530
- * training data, or a reward a trainer consumes, uses these.
1531
- * - `observedScore` — RAW. Analysis, reporting, and reward-hack DETECTION
1532
- * need the ungated number; that is how a gamed run is visible at all.
1533
- *
1534
- * A leaf module on purpose: it imports only the `RunRecord` type, so gate and
1535
- * reporting code can depend on it without pulling in the trace store that
1536
- * `mint.ts` needs.
1537
- */
1538
-
1539
- /**
1540
- * Which split's score wins when a record carries both. `'holdout'` is the
1541
- * canonical "real signal" default; `'search'` exists because some callers
1542
- * deliberately score on the search split when both are present.
1543
- */
1544
- type ScorePreference = 'holdout' | 'search';
1545
- /** Only the outcome is read, so every accessor here accepts anything carrying one. */
1546
- type Scored = Pick<RunRecord, 'outcome'>;
1547
- /** True when the authenticity gate flagged the run as gamed (`realness.gated`). */
1548
- declare function isRealnessGated(record: Scored): boolean;
1549
- /**
1550
- * The RAW score recorded on ONE split, with no cross-split fallback and no
1551
- * anti-Goodhart gate.
1552
- *
1553
- * The narrowest of the three raw readers, and the one every split-scoped
1554
- * consumer wants: a per-split report, a promotion gate, or a paired comparison
1555
- * asks "what did this run score on the split I am summarising", and answering
1556
- * it with the other split's number silently mixes populations. `undefined` =
1557
- * that split was never scored.
1558
- *
1559
- * Same warning as `observedScore`: this INCLUDES runs flagged as gamed. Never
1560
- * feed it into training data.
1561
- */
1562
- declare function observedSplitScore(record: Scored, split: ScorePreference): number | undefined;
1563
- /**
1564
- * The RAW split score the run carries, with NO anti-Goodhart gate applied.
1565
- *
1566
- * INCLUDES RUNS FLAGGED AS GAMED (`outcome.realness.gated === true`); NEVER
1567
- * feed this into training data — a fine-tune that sees it learns from gamed
1568
- * successes. It is exported anyway because analysis, reporting, and
1569
- * reward-hacking detection legitimately need the ungated number: forcing a
1570
- * gamed run to 0 collapses the proxy signal toward ground truth and makes a
1571
- * detector report "clean" on exactly the population that is being gamed.
1572
- *
1573
- * Returns `undefined` when the record carries neither score — an unscored run
1574
- * is a labeled gap, not a measured zero, and each caller picks its own
1575
- * sentinel (`?? 0`, `?? null`, skip, throw). Non-finite values are returned
1576
- * as-is; callers that care keep their own `Number.isFinite` guard.
1577
- */
1578
- declare function observedScore(record: Scored, prefer?: ScorePreference): number | undefined;
1579
- /** Which split actually carried the score, or that none did. */
1580
- type ScoreOrigin = 'holdout' | 'search' | 'unscored';
1581
- /**
1582
- * Where `observedScore` / `trainingScore` read their number from — the
1583
- * provenance label a rollout line's `reward_source` is built from, and the
1584
- * only supported way to ask "was this run scored at all" without respelling
1585
- * the field access.
1586
- */
1587
- declare function scoreOrigin(record: Scored, prefer?: ScorePreference): ScoreOrigin;
1588
- /**
1589
- * The GATED score — the only derivation allowed to reach training data.
1590
- *
1591
- * A realness-gated run scores 0 no matter what it claims, so a fine-tune
1592
- * cannot learn from a gamed success. An unscored run stays `undefined` (a
1593
- * labeled gap), keeping "we never measured this" distinct from "we measured
1594
- * zero"; callers that need a number apply their own sentinel.
1595
- */
1596
- declare function trainingScore(record: Scored, prefer?: ScorePreference): number | undefined;
1597
- /**
1598
- * `{reward, gated}` as written onto a minted `RolloutLine` — `trainingScore`
1599
- * plus the flag itself, so the gate travels into the exported row and a
1600
- * downstream filter can drop or down-weight the line.
1601
- *
1602
- * An unscored record yields `reward: null`, matching the schema's "no verdict
1603
- * exists — a labeled gap, never 0" rule. It previously collapsed to 0, which
1604
- * made a run nobody graded indistinguishable from one graded as a total
1605
- * failure, and taught any trainer reading the row that the trajectory was bad.
1606
- * A gated run still yields 0, because that IS a verdict: the gate decided.
1607
- */
1608
- declare function trainingReward(record: Scored): {
1609
- reward: number | null;
1610
- gated: boolean;
1611
- };
1612
-
1613
- /**
1614
- * Rollout minting — `tangle.rollout.v1` lines joined from the records the
1615
- * substrate ALREADY keeps. There is no separate rollout store: a rollout
1616
- * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with
1617
- * its trace (spans share `runId`), projected into the canonical line.
1618
- *
1619
- * Composition, not duplication:
1620
- * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)
1621
- * - step structure → `buildTrajectory` over the shared TraceStore
1622
- * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
1623
- * - PRM / reward-model → `reward-model-export.ts`
1624
- *
1625
- * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is
1626
- * never exported with a positive reward OR with any of the numbers that reward
1627
- * was computed from. The gate travels into the training data (`reward` forced
1628
- * to 0, `realness_gated: true`) and the whole outcome is transformed by
1629
- * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and
1630
- * `verdict` to `provenance.gated_evidence`. Mint returns
1631
- * `MintedRolloutLine[]`: the brand the training exporters require, which only
1632
- * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.
1633
- *
1634
- * A record carrying NEITHER split score is REJECTED (`ValidationError`), never
1635
- * minted at 0 — "nobody graded this" is not the same claim as "graded a total
1636
- * failure", and a trainer reading 0 learns the second. Lines that already
1637
- * carry `reward: null` (interchange imports, existing ledgers) remain valid on
1638
- * the wire; only the RunRecord→line door refuses.
1639
- *
1640
- * Records without spans become labeled GAP LINES (messages: [],
1641
- * provenance.gap) — present in the output AND surfaced in
1642
- * `missingTraces`; a capture gap is a finding, never a silent omission.
1643
- */
1644
-
1645
- /** Redactor applied to every exported string (secrets, PII). Identity by default. */
1646
- type RolloutScrubber = (text: string) => string;
1647
- interface MintRolloutOptions {
1648
- scrub?: RolloutScrubber;
1649
- /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */
1650
- maxSteps?: number;
1651
- /** Role recorded on every minted line. Default 'agent' (a solo eval run). */
1652
- role?: RolloutRole;
1653
- /** Task suite label. Default: the record's `experimentId`. */
1654
- suite?: string;
1655
- /** Injected clock for deterministic output. */
1656
- now?: () => Date;
1657
- }
1658
- interface MintRolloutResult {
1659
- rows: MintedRolloutLine[];
1660
- /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
1661
- missingTraces: string[];
1662
- }
1663
-
1664
- /**
1665
- * Join RunRecords with their traces into canonical rollout lines. Records
1666
- * without spans are emitted as labeled gap lines and reported in
1667
- * `missingTraces`. Execution-only records without a task score are rejected
1668
- * because a missing training label is not a zero reward.
1669
- */
1670
- declare function mintRolloutRows(records: RunRecord[], store: TraceStore, options?: MintRolloutOptions): Promise<MintRolloutResult>;
1671
-
1672
- /**
1673
- * Backfill reader over Claude Code project transcripts
1674
- * (~/.claude/projects/<cwd-slug>/<sessionId>.jsonl) → canonical
1675
- * chat-with-tools messages plus per-session token usage.
1676
- *
1677
- * Transcript lines consumed: type:"user" (string content or content blocks —
1678
- * text + tool_result) and type:"assistant" (content blocks — thinking, text,
1679
- * tool_use; message.usage carries tokens). Sidechain lines (isSidechain=true,
1680
- * subagent threads) are separate invocations and are excluded from the main
1681
- * transcript. Everything else (queue-operation, attachment, last-prompt…) is
1682
- * transport metadata, not conversation.
1683
- */
1684
-
1685
- declare const DEFAULT_CLAUDE_PROJECTS_DIR: string;
1686
- /** Claude Code's project-directory slug for a working directory. */
1687
- declare function claudeProjectSlug(cwd: string): string;
1688
- interface ClaudeTranscriptRef {
1689
- sessionId: string;
1690
- path: string;
1691
- }
1692
- /** Transcript files recorded for sessions launched from `cwd`. */
1693
- declare function findClaudeTranscripts(cwd: string, projectsDir?: string): Promise<ClaudeTranscriptRef[]>;
1694
- interface ClaudeUsageTotals {
1695
- tokensIn: number;
1696
- tokensOut: number;
1697
- cacheRead: number;
1698
- cacheWrite: number;
1699
- }
1700
- interface ClaudeTranscript {
1701
- messages: ChatMessage[];
1702
- usage: ClaudeUsageTotals;
1703
- /** Timestamp of the first conversation line; null = empty transcript. */
1704
- startedAt: string | null;
1705
- endedAt: string | null;
1706
- model: string | null;
1707
- }
1708
- interface ReadClaudeTranscriptOptions {
1709
- /**
1710
- * Read the sidechain (subagent) thread instead of skipping it. Subagent
1711
- * transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain
1712
- * lines end to end, so their usage is invisible without this.
1713
- */
1714
- readonly includeSidechain?: boolean;
1715
- }
1716
- /** Parse one transcript jsonl into canonical messages + usage totals. */
1717
- declare function readClaudeTranscript(path: string, options?: ReadClaudeTranscriptOptions): Promise<ClaudeTranscript>;
1718
-
1719
- /**
1720
- * Read-only backfill reader over the opencode sqlite store
1721
- * (~/.local/share/opencode/opencode.db) → canonical chat-with-tools messages.
1722
- *
1723
- * Schema consumed (observed, 2026-07): `session` rows carry directory /
1724
- * parent_id / agent / model / cost / tokens_*; `message` rows carry a JSON
1725
- * `data` blob ({role, modelID, providerID, tokens, cost, finish}); `part`
1726
- * rows carry the actual content ({type: text|reasoning|tool|step-start|
1727
- * step-finish|snapshot…}). Tool parts hold {callID, state:{input, output,
1728
- * status}} — both the call and its result, which we split into an assistant
1729
- * tool_call plus a role:"tool" result message.
1730
- *
1731
- * The store is mutable and can be corrupt (a `.corrupt-bak` sibling ships
1732
- * next to it in the wild), so `openOpencodeDb` returns null instead of
1733
- * throwing — callers record a gap line, never crash the backfill.
1734
- */
1735
-
1736
- declare const DEFAULT_OPENCODE_DB: string;
1737
- interface OpencodeSessionRow {
1738
- id: string;
1739
- parentId: string | null;
1740
- directory: string;
1741
- agent: string | null;
1742
- /** Raw session.model JSON: {id, providerID, variant} where present. */
1743
- model: {
1744
- id?: string;
1745
- providerID?: string;
1746
- } | null;
1747
- costUsd: number;
1748
- tokensInput: number;
1749
- tokensOutput: number;
1750
- tokensReasoning: number;
1751
- tokensCacheRead: number;
1752
- tokensCacheWrite: number;
1753
- timeCreated: number;
1754
- timeUpdated: number;
1755
- }
1756
- /** Open the store read-only; null = unavailable/corrupt (caller records a gap). */
1757
- declare function openOpencodeDb(path?: string): Promise<DatabaseSync | null>;
1758
- /** Sessions whose cwd is `directory` (the worker-clone join key). */
1759
- declare function findOpencodeSessionsByDirectory(db: DatabaseSync, directory: string): OpencodeSessionRow[];
1760
- declare function findOpencodeSessionById(db: DatabaseSync, sessionId: string): OpencodeSessionRow | null;
1761
- /**
1762
- * Convert one session's message+part rows into canonical messages.
1763
- * An opencode assistant message row spans several model steps; each step's
1764
- * parts (reasoning → text → tool …) become one assistant message followed by
1765
- * the role:"tool" results of its calls, preserving order.
1766
- */
1767
- declare function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[];
1768
-
1769
- /**
1770
- * Per-format accounting of the anti-Goodhart gate for one dataset release.
1771
- *
1772
- * The defect this exists to make impossible: a dataset card that STATES what
1773
- * the gate does while the build does something else. A sentence in a README is
1774
- * a claim about bytes it never reads, so it drifts the moment an exporter
1775
- * changes — and the drift ships to whoever downloads the dataset.
1776
- *
1777
- * So the card is not allowed to assert anything about the gate. The build
1778
- * measures the rows it is ABOUT TO WRITE (`measureFormatGate`), the measurement
1779
- * is checked against the declared per-format disposition (`assertGateReport`,
1780
- * which throws rather than warns), and the card renders only numbers handed to
1781
- * it. A card that disagrees with its own data files cannot be produced without
1782
- * failing the build first.
1783
- *
1784
- * The dispositions themselves are the release policy, stated once as data:
1785
- *
1786
- * sft EXCLUDE — an SFT row is an imitation target. A gamed
1787
- * trajectory must never be imitated, at any weight.
1788
- * verifiers ZERO_AND_FLAG — reward is a signed learning signal here, so a
1789
- * gamed trajectory at reward 0 is a correct
1790
- * negative. Dropping it would bias the negative
1791
- * population toward honest failures and leave a
1792
- * trainer no example of what gaming looks like
1793
- * when it is penalized.
1794
- * rft ZERO_AND_FLAG — RFT re-samples the completion; only the prompt
1795
- * and the grader's `reference.*` verdict ship, so
1796
- * nothing gamed is imitated. The flag is what lets
1797
- * a grader author skip the instance.
1798
- * raw ZERO_AND_FLAG — a faithful audit dump. Removing rows from it
1799
- * would defeat its only purpose, and the gated
1800
- * row is the one an auditor most wants.
1801
- *
1802
- * `ZERO_AND_FLAG` is never `reward: 0` alone. Zeroing without the label makes a
1803
- * faked success indistinguishable from an honest failure — it hides the gamed
1804
- * population from the buyer instead of disclosing it. Every included format
1805
- * carries `realness_gated` on the row itself.
1806
- *
1807
- * And `ZERO_AND_FLAG` means the whole outcome, not the scalar. A gated row that
1808
- * ships `reward: 0` beside the per-layer verifier scores the reward was
1809
- * computed from has not been zeroed in any sense a trainer respects; the
1810
- * accounting therefore measures every reward-derived number each format writes,
1811
- * not just the one field.
1812
- */
1813
-
1814
- /** What a format does with a line the realness gate flagged. */
1815
- type GateDisposition = 'exclude' | 'zero-and-flag';
1816
- declare const FORMAT_GATE_DISPOSITION: Record<ReleaseFormat, GateDisposition>;
1817
- /** What the gate accounting reads off an emitted row, per format. */
1818
- interface ReleaseRowRef {
1819
- rollout_id: string;
1820
- reward: number | null;
1821
- /**
1822
- * The rest of the row that was DERIVED from the reward — the per-layer score
1823
- * dict, the judge verdict record, whatever this format ships beside the
1824
- * scalar. Walked for positive numbers, so the certification is about the
1825
- * whole outcome rather than one field.
1826
- *
1827
- * Absent when the format's row carries nothing but the scalar. NOT the whole
1828
- * row: `cost.tokens_in`, `wall_s` and `total_steps` are positive numbers that
1829
- * have nothing to do with the reward, and a certification that flags them is
1830
- * a certification nobody can act on.
1831
- */
1832
- evidence?: unknown;
1833
- /**
1834
- * The screen claim AS EMITTED — read off the row, not off the line it came
1835
- * from, because what ships is what matters. Required, not optional: an
1836
- * optional field is how a format quietly opts out of the check that reads it,
1837
- * and every emitted row shape carries `RealnessLabels` precisely so no adapter
1838
- * has to.
1839
- */
1840
- realness_screened: boolean | null;
1841
- /**
1842
- * The part of an emitted `steps[]` the wire format does not declare.
1843
- *
1844
- * Separate from `evidence` because the declared step fields are FULL of
1845
- * legitimate positive numbers — `durationMs`, `llm_call_count`,
1846
- * `prompt_token_ids` — and a certification that flags those is one nobody can
1847
- * act on. Only the undeclared remainder is unclassified reward-bearing
1848
- * payload, which is the same partition the check applies.
1849
- *
1850
- * Set only by formats whose row carries steps: today `raw` alone.
1851
- */
1852
- stepEvidence?: unknown;
1853
- }
1854
- /** A positive number found inside an emitted gated row, with where it was. */
1855
- interface EmittedEvidence {
1856
- /** JSON-ish path from the row's evidence root, e.g. `metrics['layer.tests']`. */
1857
- path: string;
1858
- value: number;
1859
- }
1860
- interface FormatGateCounts {
1861
- /** Gated lines that reached this format's exporter. */
1862
- input: number;
1863
- /** Gated rows the format actually wrote. */
1864
- emitted: number;
1865
- /**
1866
- * Gated lines this format did not write. Not all of these are the gate:
1867
- * `verifiers` also drops gap lines (empty transcript) and `rft` drops lines
1868
- * with no prompt turn, so an excluded count can mix both causes.
1869
- */
1870
- excluded: number;
1871
- /** Highest reward on an emitted gated row; `null` when none was emitted. */
1872
- maxEmittedReward: number | null;
1873
- /**
1874
- * The largest positive number found in the reward-DERIVED payload of an
1875
- * emitted gated row, and its path; `null` when there is none.
1876
- *
1877
- * This column exists because the release once certified CLEAN while leaking.
1878
- * `assertGateReport` inspected `outcome.reward` alone, so a gated row shipping
1879
- * `reward: 0` next to `metrics['layer.tests']: 1` — the deterministic verifier
1880
- * score the reward was computed from, and the per-rubric score dict of the
1881
- * Prime Intellect verifiers format — passed, and the card rendered "max reward
1882
- * | 0" over a file that carried the gamed signal at full value. A wrong
1883
- * certification is worse than the leak: it is the leak plus a document saying
1884
- * there isn't one.
1885
- */
1886
- maxEmittedEvidence: EmittedEvidence | null;
1887
- /**
1888
- * Rows this format wrote carrying a positive reward whose producer DECLARED
1889
- * that no authenticity screen ever ran on it (`realness_screened: false`).
1890
- *
1891
- * Measured over EVERY emitted row, not just the gated ones: an unscreened
1892
- * reward is by definition one the gate never had a verdict on, so it is not in
1893
- * the gated set and a measurement scoped to that set would report 0 forever.
1894
- * `assertMinted` already refuses these, which is exactly why the release still
1895
- * measures them — the last door before a public dataset does not get to assume
1896
- * the earlier doors held.
1897
- */
1898
- unscreenedPositiveRows: number;
1899
- /** Highest reward on such a row; `null` when there is none. */
1900
- maxUnscreenedReward: number | null;
1901
- /**
1902
- * The largest positive number found in an emitted gated row's UNDECLARED
1903
- * per-step payload, and its path; `null` when there is none.
1904
- *
1905
- * The column exists because the gate read `outcome` and nothing else for
1906
- * three rounds, so a gated line shipping `steps: [{kind, name, reward: 0.86}]`
1907
- * certified clean — the release accounting agreed with the exporter that a
1908
- * per-step reward was not a reward.
1909
- */
1910
- maxEmittedStepEvidence: EmittedEvidence | null;
1911
- }
1912
- interface GateReport {
1913
- /** Gated lines in the release input, after the split/proposer filters. */
1914
- gatedLines: number;
1915
- byFormat: Partial<Record<ReleaseFormat, FormatGateCounts>>;
1916
- }
1917
- /** Rollout ids of every gated line, the key the emitted rows are matched on. */
1918
- declare function gatedRolloutIds(lines: readonly MintedRolloutLine[]): Set<string>;
1919
- /**
1920
- * Row refs per format. Written as one adapter per format so that the knowledge
1921
- * of WHERE the id and reward live in each published shape sits next to the
1922
- * assertion that uses it — an exporter that moves either field breaks here
1923
- * rather than silently reporting zero gated rows.
1924
- */
1925
- declare const releaseRowRefs: {
1926
- sft: (rows: readonly SftRow[]) => ReleaseRowRef[];
1927
- verifiers: (rows: readonly VerifiersRolloutOutput[]) => ReleaseRowRef[];
1928
- rft: (rows: readonly RftItem[]) => ReleaseRowRef[];
1929
- raw: (lines: readonly MintedRolloutLine[]) => ReleaseRowRef[];
1930
- };
1931
- /** Measure one format's gated rows from the refs of the rows about to be written. */
1932
- declare function measureFormatGate(gated: ReadonlySet<string>, refs: readonly ReleaseRowRef[]): FormatGateCounts;
1933
- /**
1934
- * Fail the build when the measurement disagrees with the declared policy.
1935
- *
1936
- * Throws, never filters: an emitted positive reward on a gated row means an
1937
- * exporter upstream stopped applying the gate, and silently dropping the row
1938
- * would hide the producer that made it — the producer is the actual defect.
1939
- *
1940
- * Certifies the whole emitted outcome, not `reward` alone. The earlier version
1941
- * checked one field and therefore certified a release CLEAN while its
1942
- * `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in
1943
- * the top-level `metrics` dict — the card then rendered "max reward | 0" over
1944
- * exactly that file. A certification that is wrong is worse than an
1945
- * uncertified leak, so the checks it runs are no longer written down here at
1946
- * all: it iterates `GATE_CHECK_IDS` under its own declared policy.
1947
- */
1948
- declare function assertGateReport(report: GateReport): void;
1949
-
1950
- /**
1951
- * Deterministic scrubbing pass over rollout-ledger lines before public release.
1952
- *
1953
- * Every rule is a pure regex rewrite applied to every string value in a line
1954
- * (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed
1955
- * line is still a valid `tangle.rollout.v1` line. Rules are idempotent:
1956
- * scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is
1957
- * the property the release pipeline relies on to prove nothing half-scrubbed
1958
- * ships. Rule order matters: whole `KEY=value` env pairs are redacted before
1959
- * the bare-key rule so one secret is never counted twice.
1960
- */
1961
-
1962
- interface ScrubRule {
1963
- name: string;
1964
- pattern: RegExp;
1965
- /** Rewrite for one match; `g1` is the first capture group when present. */
1966
- rewrite: (match: string, g1?: string) => string;
1967
- }
1968
- declare const SCRUB_RULES: readonly ScrubRule[];
1969
- /** Rule name → number of matches rewritten. Always carries every rule (0 is data). */
1970
- type ScrubCounts = Record<string, number>;
1971
- declare function emptyScrubCounts(): ScrubCounts;
1972
- declare function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts;
1973
- declare function scrubText(text: string, counts: ScrubCounts): string;
1974
- /**
1975
- * Scrub every string value in a line; structure and key order are preserved.
1976
- *
1977
- * `assertMinted` on the way out rather than a cast: scrubbing rebuilds the
1978
- * object, so the brand has to be re-earned, and re-validating proves the rules
1979
- * did not rewrite a field the schema constrains (`reward` is a number, not a
1980
- * string, so no rule should ever touch it — this is what checks that).
1981
- */
1982
- declare function scrubRolloutLine(line: MintedRolloutLine, counts: ScrubCounts): MintedRolloutLine;
1983
- declare function scrubLines(lines: MintedRolloutLine[]): {
1984
- lines: MintedRolloutLine[];
1985
- counts: ScrubCounts;
1986
- };
1987
- /**
1988
- * A `RolloutScrubber` (text → text) applying the full rule set — the
1989
- * default hook to pass to `mintRolloutRows({ scrub })` so lines are
1990
- * scrubbed at mint time, before they ever reach a ledger file. Release
1991
- * builds re-run `scrubLines` regardless (idempotent), so double-scrubbing
1992
- * is safe and counted as zero.
1993
- */
1994
- declare function defaultRolloutScrubber(text: string): string;
1995
-
1996
- /**
1997
- * HuggingFace dataset-card (README.md) generation for a rollout-ledger release.
1998
- *
1999
- * The card is a pure function of the SCRUBBED lines plus the release options —
2000
- * no timestamps, no environment reads — so rebuilding from the same ledger
2001
- * yields byte-identical output. It documents the schema, provenance (run ids,
2002
- * generations, the official judge), per-role reward semantics including the
2003
- * inherited/contribution caveat, and a role × reward counts table.
2004
- */
2005
-
2006
- declare const RELEASE_FORMATS: readonly ["sft", "verifiers", "rft", "raw"];
2007
- type ReleaseFormat = (typeof RELEASE_FORMATS)[number];
2008
- /** Format → data file path inside the dataset dir (train split only). */
2009
- declare const FORMAT_FILES: Record<ReleaseFormat, string>;
2010
- interface DatasetCardInputs {
2011
- /** Scrubbed, release-filtered lines (what actually ships). */
2012
- lines: MintedRolloutLine[];
2013
- formats: ReleaseFormat[];
2014
- includeProposers: boolean;
2015
- /** Source ledger basenames, for provenance. */
2016
- sourceFiles: string[];
2017
- scrubTotals: ScrubCounts;
2018
- excluded: {
2019
- proposers: number;
2020
- nonTrain: number;
2021
- };
2022
- formatCounts: Partial<Record<ReleaseFormat, number>>;
2023
- /**
2024
- * Per-format anti-Goodhart accounting MEASURED on the rows the build wrote.
2025
- * Required, not optional: the card's only statement about the gate is a
2026
- * render of these numbers, so a card cannot be produced without them and
2027
- * cannot drift from the data files it ships beside.
2028
- */
2029
- gate: GateReport;
2030
- }
2031
- declare function buildDatasetCard(inputs: DatasetCardInputs): string;
2032
-
2033
- /**
2034
- * One-command HuggingFace dataset release from rollout ledgers:
2035
- *
2036
- * agent-eval rollout-release <ledger.jsonl...> --out <dir> \
2037
- * [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]
2038
- *
2039
- * Pipeline per input ledger: read + validate → fail-closed filters
2040
- * (trainable split only; proposer sessions dropped unless
2041
- * --include-proposers, they contain improvement-loop harness source) →
2042
- * deterministic scrub → export the requested formats + scrub-report.json +
2043
- * auto-generated README.md card. Deterministic: same inputs and flags →
2044
- * byte-identical output dir.
2045
- *
2046
- * --push uploads the built dir with `huggingface-cli upload` only when the
2047
- * CLI exists on PATH and HF_TOKEN is present in the env; the token is
2048
- * never printed. Everything else runs fully offline.
2049
- */
2050
-
2051
- interface BuildOptions {
2052
- out: string;
2053
- formats: ReleaseFormat[];
2054
- includeProposers: boolean;
2055
- }
2056
- interface ScrubReport {
2057
- /** Input ledger path → rule → rewrite count (only shipped lines are scrubbed). */
2058
- files: Record<string, ScrubCounts>;
2059
- totals: ScrubCounts;
2060
- excluded: {
2061
- proposers: number;
2062
- nonTrain: number;
2063
- };
2064
- }
2065
- interface BuildSummary {
2066
- inputs: string[];
2067
- read: number;
2068
- kept: number;
2069
- scrub: ScrubReport;
2070
- formatCounts: Partial<Record<ReleaseFormat, number>>;
2071
- /** Per-format anti-Goodhart accounting, measured on the rows written. */
2072
- gate: GateReport;
2073
- files: string[];
2074
- }
2075
- declare function buildHfDataset(inputs: string[], options: BuildOptions): Promise<BuildSummary>;
2076
- declare function planPushCommand(repo: string, outDir: string): string[];
2077
- declare function pushDataset(repo: string, outDir: string): void;
2078
- interface RolloutReleaseCliArgs extends BuildOptions {
2079
- inputs: string[];
2080
- push: string | null;
2081
- }
2082
- declare const ROLLOUT_RELEASE_USAGE = "usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]";
2083
- declare function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs;
2084
- /** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
2085
- declare function runRolloutReleaseCli(argv: string[]): Promise<number>;
2086
-
2087
- export { ATIF_SCHEMA_VERSION, type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, type EmittedEvidence, FORMAT_FILES, FORMAT_GATE_DISPOSITION, type FormatGateCounts, type FromHarborOptions, GATE_CHECKS, GATE_CHECK_IDS, GATE_POLICIES, type GateCheck, type GateCheckDisposition, type GateCheckId, type GateCheckedOutcome, type GateDisposition, type GateEntryPoint, type GatePolicy, type GateReport, type GatedEvidence, HARBOR_IMPORT_GAP, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type RealnessLabels, type ReleaseFormat, type ReleaseRowRef, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScoreOrigin, type ScorePreference, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertGateReport, assertMinted, assertMintedLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, fromHarborTrajectory, gateErrors, gateGamedOutcome, gatedEvidenceOf, gatedRolloutIds, isRealnessGated, isRolloutLine, isTrainableSplit, measureFormatGate, mintRolloutRows, observedScore, observedSplitScore, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutJournal, readRolloutLedger, realnessLabels, relabelImportedSplit, releaseRowRefs, runRolloutReleaseCli, scoreOrigin, scrubLines, scrubRolloutLine, scrubText, toHarborTrajectories, toHarborTrajectory, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, trainingReward, trainingScore, validateRolloutLine, writeRolloutLedger };
1
+ import { A as isTrainableSplit, C as TRAINABLE_SPLITS, D as assertRolloutLine, E as assertMintedLines, O as gateGamedOutcome, S as RolloutTask, T as assertMinted, _ as RolloutPolicy, a as GatedEvidence, b as RolloutSplit, c as ROLLOUT_CAPTURES, d as ROLLOUT_SPLITS, f as RolloutArtifacts, g as RolloutOutcome, h as RolloutLine, i as ChatToolCall, j as validateRolloutLine, k as isRolloutLine, l as ROLLOUT_ROLES, m as RolloutCostBlock, n as ChatMessage, o as MintedRolloutLine, p as RolloutCapture, r as ChatRole, s as MintedRolloutOutcome, t as CHAT_ROLES, u as ROLLOUT_SCHEMA, v as RolloutProvenance, w as ToolDef, x as RolloutStep, y as RolloutRole } from "../schema-Cef2cFmb.js";
2
+ import { $ as ScorePreference, $t as toVerifiersRolloutOutput, A as ReleaseRowRef, At as GATE_CHECK_IDS, B as readOpencodeSessionMessages, Bt as RealnessLabels, C as scrubRolloutLine, Ct as HarborToolCall, D as FormatGateCounts, Dt as toHarborTrajectories, E as FORMAT_GATE_DISPOSITION, Et as relabelImportedSplit, F as DEFAULT_OPENCODE_DB, Ft as GateCheckedOutcome, G as claudeProjectSlug, Gt as VerifiersRolloutOutput, H as ClaudeTranscriptRef, Ht as RftItem, I as OpencodeSessionRow, It as GateEntryPoint, J as MintRolloutOptions, Jt as toJsonl, K as findClaudeTranscripts, Kt as VerifiersTokenUsage, L as findOpencodeSessionById, Lt as GatePolicy, M as gatedRolloutIds, Mt as GateCheck, N as measureFormatGate, Nt as GateCheckDisposition, O as GateDisposition, Ot as toHarborTrajectory, P as releaseRowRefs, Pt as GateCheckId, Q as ScoreOrigin, Qt as toSftRows, R as findOpencodeSessionsByDirectory, Rt as gateErrors, S as scrubLines, St as HarborSubagentTrajectoryRef, T as EmittedEvidence, Tt as fromHarborTrajectory, U as ClaudeUsageTotals, Ut as SftExportOptions, V as ClaudeTranscript, Vt as RewardRow, W as DEFAULT_CLAUDE_PROJECTS_DIR, Wt as SftRow, X as RolloutScrubber, Xt as toRftItem, Y as MintRolloutResult, Yt as toRewardRows, Z as mintRolloutRows, Zt as toRftItems, _ as ScrubCounts, _t as HarborMetrics, a as ScrubReport, at as trainingScore, b as defaultRolloutScrubber, bt as HarborStep, c as planPushCommand, ct as readRolloutLedger, d as DatasetCardInputs, dt as FromHarborOptions, en as toVerifiersRolloutOutputs, et as isRealnessGated, f as FORMAT_FILES, ft as HARBOR_IMPORT_GAP, g as SCRUB_RULES, gt as HarborImageSource, h as buildDatasetCard, ht as HarborFinalMetrics, i as RolloutReleaseCliArgs, it as trainingReward, j as assertGateReport, jt as GATE_POLICIES, k as GateReport, kt as GATE_CHECKS, l as pushDataset, lt as writeRolloutLedger, m as ReleaseFormat, mt as HarborContentPart, n as BuildSummary, nt as observedSplitScore, o as buildHfDataset, ot as appendRolloutLines, p as RELEASE_FORMATS, pt as HarborAgent, q as readClaudeTranscript, qt as realnessLabels, r as ROLLOUT_RELEASE_USAGE, rt as scoreOrigin, s as parseRolloutReleaseArgs, st as readRolloutJournal, t as BuildOptions, tt as observedScore, u as runRolloutReleaseCli, ut as ATIF_SCHEMA_VERSION, v as ScrubRule, vt as HarborObservation, w as scrubText, wt as HarborTrajectory, x as emptyScrubCounts, xt as HarborStepSource, y as addScrubCounts, yt as HarborObservationResult, z as openOpencodeDb, zt as gatedEvidenceOf } from "../index-2JJSA6-r2.js";
3
+ export { ATIF_SCHEMA_VERSION, type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, type EmittedEvidence, FORMAT_FILES, FORMAT_GATE_DISPOSITION, type FormatGateCounts, type FromHarborOptions, GATE_CHECKS, GATE_CHECK_IDS, GATE_POLICIES, type GateCheck, type GateCheckDisposition, type GateCheckId, type GateCheckedOutcome, type GateDisposition, type GateEntryPoint, type GatePolicy, type GateReport, type GatedEvidence, HARBOR_IMPORT_GAP, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type RealnessLabels, type ReleaseFormat, type ReleaseRowRef, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScoreOrigin, type ScorePreference, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertGateReport, assertMinted, assertMintedLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, fromHarborTrajectory, gateErrors, gateGamedOutcome, gatedEvidenceOf, gatedRolloutIds, isRealnessGated, isRolloutLine, isTrainableSplit, measureFormatGate, mintRolloutRows, observedScore, observedSplitScore, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutJournal, readRolloutLedger, realnessLabels, relabelImportedSplit, releaseRowRefs, runRolloutReleaseCli, scoreOrigin, scrubLines, scrubRolloutLine, scrubText, toHarborTrajectories, toHarborTrajectory, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, trainingReward, trainingScore, validateRolloutLine, writeRolloutLedger };