@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -1,1048 +1,3 @@
1
- import { DatabaseSync } from 'node:sqlite';
2
-
3
- /**
4
- * `tangle.rollout.v1` — THE canonical rollout serialization, owned by
5
- * agent-eval. One JSONL line per agent invocation (a solo eval run, a
6
- * supervisor episode, a worker session, a proposer shot, a judge call, an
7
- * analyst pass), labeled with its task/split coordinates and a single
8
- * scalar reward, carrying the FULL message transcript inline.
9
- *
10
- * This schema is the reconciliation of two prior producers:
11
- * - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
12
- * provenance hashes, the realness gate travelling into the reward,
13
- * trace-derived steps.
14
- * - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
15
- * role, task.split/rep, parent_rollout_id, policy provenance, capture
16
- * provenance, inline canonical chat-with-tools messages.
17
- * Where the two conflicted, RunRecord-derived semantics won; the wire
18
- * field names follow the ledger (snake_case). See `docs/rollout.md` for
19
- * the field-by-field decision table.
20
- *
21
- * Messages are inlined — never referenced — because every harness store a
22
- * rollout can be recovered from is mutable or garbage-collected. A line
23
- * must stay a complete training/eval example on its own.
24
- *
25
- * `outcome.reward` is THE single scalar (null = no verdict exists — a
26
- * labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
27
- * flag: a gated line must never export as a positive training example.
28
- */
29
- declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
30
- /** `agent` = a solo evaluation run (no multi-agent topology). */
31
- type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst';
32
- declare const ROLLOUT_ROLES: readonly RolloutRole[];
33
- /** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
34
- type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
35
- declare const ROLLOUT_SPLITS: readonly RolloutSplit[];
36
- /** Splits that may ship in training exports. Everything else is fail-closed excluded. */
37
- declare const TRAINABLE_SPLITS: readonly RolloutSplit[];
38
- declare function isTrainableSplit(split: RolloutSplit): boolean;
39
- /** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
40
- type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
41
- declare const ROLLOUT_CAPTURES: readonly RolloutCapture[];
42
- type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
43
- declare const CHAT_ROLES: readonly ChatRole[];
44
- interface ChatToolCall {
45
- id: string;
46
- type: 'function';
47
- function: {
48
- name: string;
49
- /** JSON-encoded argument object, exactly as the model emitted it. */
50
- arguments: string;
51
- };
52
- }
53
- interface ChatMessage {
54
- role: ChatRole;
55
- content: string | null;
56
- /** Reasoning/thinking channel where the harness captured it (full fidelity). */
57
- reasoning_content?: string;
58
- tool_calls?: ChatToolCall[];
59
- /** Required on role:"tool" — the ChatToolCall this result answers. */
60
- tool_call_id?: string;
61
- name?: string;
62
- }
63
- interface ToolDef {
64
- type: 'function';
65
- function: {
66
- name: string;
67
- description?: string;
68
- parameters?: Record<string, unknown>;
69
- };
70
- }
71
- /**
72
- * Compact trace-span projection (llm/tool step) carried alongside the
73
- * conversation when the line was minted from a trace. Optional: lines
74
- * recovered from harness stores have no span structure.
75
- */
76
- interface RolloutStep {
77
- kind: string;
78
- name: string;
79
- /** llm: last-message summary · tool: stringified args. Scrubbed. */
80
- input?: string;
81
- /** llm: output text · tool: stringified result. Scrubbed. */
82
- output?: string;
83
- status?: 'ok' | 'error';
84
- durationMs?: number;
85
- }
86
- interface RolloutTask {
87
- /** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
88
- suite: string;
89
- instance_id: string;
90
- split: RolloutSplit;
91
- /** Sampling seed the campaign pinned; null = not recorded. */
92
- seed: number | null;
93
- /** Replicate index (0-based). */
94
- rep: number;
95
- }
96
- interface RolloutPolicy {
97
- /** Harness that drove the invocation (e.g. "opencode", "claude", "pi-loops"). */
98
- harness: string | null;
99
- harness_version: string | null;
100
- model: string | null;
101
- provider: string | null;
102
- /** Commit of the agent profile / candidate under evaluation. */
103
- profile_commit: string | null;
104
- /** sha256 of the effective prompt (post-steering), when recorded. */
105
- prompt_hash?: string | null;
106
- /** sha256 of the effective run config, when recorded. */
107
- config_hash?: string | null;
108
- /** Canonical agent-profile cell identity, when the run carries one. */
109
- agent_profile_cell_id?: string | null;
110
- /** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */
111
- sampling: Record<string, unknown> | null;
112
- }
113
- interface RolloutOutcome {
114
- /**
115
- * THE single scalar training signal — the official verdict.
116
- * null = no verdict exists for this invocation (a labeled gap, never 0).
117
- */
118
- reward: number | null;
119
- /** Where the reward came from (judge id; "/inherited" = parent episode's). */
120
- reward_source: string | null;
121
- /** Raw judge verdict record, verbatim. */
122
- verdict: unknown;
123
- /** Everything that is NOT the scalar reward. */
124
- metrics: Record<string, unknown>;
125
- is_completed: boolean;
126
- is_truncated: boolean;
127
- error: string | null;
128
- /**
129
- * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run
130
- * faked its success signal. Reward is forced to 0 at mint time and the
131
- * line never qualifies for SFT.
132
- */
133
- realness_gated: boolean;
134
- }
135
- interface RolloutCostBlock {
136
- usd: number | null;
137
- tokens_in: number | null;
138
- tokens_out: number | null;
139
- tokens_reasoning: number | null;
140
- cache_read: number | null;
141
- cache_write: number | null;
142
- wall_s: number | null;
143
- }
144
- interface RolloutArtifacts {
145
- patch_path: string | null;
146
- run_dir: string | null;
147
- /** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
148
- transcript_ref: string | null;
149
- }
150
- interface RolloutProvenance {
151
- captured_at: string;
152
- capture: RolloutCapture;
153
- /** Present on gap lines: why `messages` could not be recovered. */
154
- gap?: string;
155
- }
156
- interface RolloutLine {
157
- schema: typeof ROLLOUT_SCHEMA;
158
- rollout_id: string;
159
- /** Spawning invocation within the same episode (worker → supervisor). */
160
- parent_rollout_id: string | null;
161
- run_id: string;
162
- /** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */
163
- experiment_id: string | null;
164
- /** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */
165
- candidate_id: string | null;
166
- /** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */
167
- generation: number | null;
168
- /** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */
169
- candidate_index: number | null;
170
- role: RolloutRole;
171
- task: RolloutTask;
172
- policy: RolloutPolicy;
173
- /** Full transcript, inline. [] = gap line (see provenance.gap). */
174
- messages: ChatMessage[];
175
- tool_defs: ToolDef[];
176
- /** Trace-span projections, when minted from a trace. */
177
- steps?: RolloutStep[];
178
- outcome: RolloutOutcome;
179
- cost: RolloutCostBlock;
180
- artifacts: RolloutArtifacts;
181
- provenance: RolloutProvenance;
182
- }
183
- declare function validateRolloutLine(value: unknown): string[];
184
- declare function assertRolloutLine(value: unknown, context?: string): asserts value is RolloutLine;
185
- declare function isRolloutLine(value: unknown): value is RolloutLine;
186
-
187
- /**
188
- * Pure exporters over `tangle.rollout.v1` lines → the training-data shapes
189
- * the improvement loops feed:
190
- * - SFT chat JSONL (clean trainable successes, {messages, metadata})
191
- * - reward rows (every scored line, success or failure, with steps)
192
- * - Prime Intellect verifiers RolloutOutput (prompt/completion split + reward)
193
- * - OpenAI RFT items (prompt turns + verdict reference fields)
194
- *
195
- * All exporters are pure functions of the lines — filtering (never train on
196
- * holdout, reward thresholds, the realness gate) happens HERE, on inline
197
- * labels, no joins.
198
- */
199
-
200
- interface TrainingExportOptions {
201
- /** Include held-out evaluation data in training output. Default false. */
202
- allowHeldOutTrainingData?: boolean;
203
- /** Require reward to be strictly greater than this value. Default 0. */
204
- minimumQualityExclusive?: number;
205
- }
206
- type SftExportOptions = TrainingExportOptions;
207
- interface SftRow {
208
- messages: ChatMessage[];
209
- metadata: {
210
- rollout_id: string;
211
- run_id: string;
212
- candidate_id: string | null;
213
- instance_id: string;
214
- reward: number;
215
- };
216
- }
217
- /**
218
- * Supervised fine-tune rows: the completed conversation of each qualifying
219
- * line. Fail-closed filters: trainable split only (never holdout/canary),
220
- * positive reward, realness-gated lines never qualify, gap lines carry
221
- * no trainable content.
222
- */
223
- declare function toSftRows(lines: RolloutLine[], options?: SftExportOptions): SftRow[];
224
- interface RewardRow {
225
- /** First user turn — the task prompt. */
226
- prompt: string;
227
- steps: RolloutStep[];
228
- reward: number;
229
- metadata: {
230
- rollout_id: string;
231
- run_id: string;
232
- candidate_id: string | null;
233
- instance_id: string;
234
- split: RolloutSplit;
235
- };
236
- }
237
- /**
238
- * Reward-labeled rows for completed, positive-quality training runs.
239
- */
240
- declare function toRewardRows(lines: RolloutLine[], options?: TrainingExportOptions): RewardRow[];
241
- interface VerifiersTokenUsage {
242
- input_tokens: number | null;
243
- output_tokens: number | null;
244
- reasoning_tokens: number | null;
245
- cache_read_tokens: number | null;
246
- cache_write_tokens: number | null;
247
- }
248
- interface VerifiersRolloutOutput {
249
- /** Messages through the last turn BEFORE the first assistant turn. */
250
- prompt: ChatMessage[];
251
- /** The first assistant turn onward — what the policy produced. */
252
- completion: ChatMessage[];
253
- reward: number | null;
254
- metrics: Record<string, unknown>;
255
- tool_defs: ToolDef[];
256
- token_usage: VerifiersTokenUsage;
257
- info: {
258
- task: RolloutLine['task'];
259
- policy: RolloutLine['policy'];
260
- rollout_id: string;
261
- run_id: string;
262
- experiment_id: string | null;
263
- candidate_id: string | null;
264
- generation: number | null;
265
- candidate_index: number | null;
266
- role: RolloutLine['role'];
267
- };
268
- }
269
- declare function toVerifiersRolloutOutput(line: RolloutLine): VerifiersRolloutOutput;
270
- declare function toVerifiersRolloutOutputs(lines: RolloutLine[], options?: TrainingExportOptions): VerifiersRolloutOutput[];
271
- interface RftItem {
272
- /** Prompt turns only — the graded completion is re-sampled during RFT. */
273
- messages: ChatMessage[];
274
- /** Verdict/label fields the grader references as item.reference.* */
275
- reference: {
276
- reward: number | null;
277
- reward_source: string | null;
278
- verdict: unknown;
279
- instance_id: string;
280
- suite: string;
281
- split: RolloutSplit;
282
- rollout_id: string;
283
- };
284
- }
285
- declare function toRftItem(line: RolloutLine): RftItem;
286
- /** RFT needs a real prompt: lines whose transcript starts with prompt turns. */
287
- declare function toRftItems(lines: RolloutLine[], options?: TrainingExportOptions): RftItem[];
288
- declare function toJsonl(rows: ReadonlyArray<unknown>): string;
289
-
290
- /**
291
- * Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`
292
- * lines. Writes validate BEFORE touching disk (a bad line never lands);
293
- * reads validate line-by-line and fail loud with the line number, because a
294
- * silently-skipped rollout is a corrupted dataset.
295
- */
296
-
297
- /** Replace the ledger file with exactly `lines`. */
298
- declare function writeRolloutLedger(path: string, lines: RolloutLine[]): Promise<void>;
299
- /** Append `lines` to the ledger file (created if absent). */
300
- declare function appendRolloutLines(path: string, lines: RolloutLine[]): Promise<void>;
301
- /**
302
- * Read and validate every line. Throws on the first malformed/invalid line
303
- * (with its 1-based line number) — fail-closed, never a silent drop.
304
- */
305
- declare function readRolloutLedger(path: string): Promise<RolloutLine[]>;
306
-
307
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
308
- type AgentProfileDimensionValue = string | number | boolean | null;
309
- interface AgentProfileSource {
310
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
311
- kind: string;
312
- /** sha256 over the canonical source profile object. */
313
- hash: string;
314
- }
315
- interface AgentProfileHarness {
316
- id: string;
317
- version?: string;
318
- hash?: string;
319
- }
320
- interface AgentProfileCell {
321
- schemaVersion: AgentProfileCellSchemaVersion;
322
- cellId: string;
323
- profileId: string;
324
- sourceProfile: AgentProfileSource;
325
- harness?: AgentProfileHarness;
326
- model?: string;
327
- promptHash?: string;
328
- dimensions?: Record<string, AgentProfileDimensionValue>;
329
- }
330
-
331
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
332
- interface BudgetSpec {
333
- tokens?: number;
334
- wallMs?: number;
335
- calls?: number;
336
- usd?: number;
337
- }
338
- interface RunOutcome$1 {
339
- score?: number;
340
- pass?: boolean;
341
- failureClass?: FailureClass;
342
- notes?: string;
343
- }
344
- /**
345
- * Layer — optional classification in a nested build workflow.
346
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
347
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
348
- * `app-runtime`: a run of the generated agent against a domain scenario.
349
- * `meta`: any meta-eval (judge replay, correlation analysis).
350
- */
351
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
352
- interface Run {
353
- runId: string;
354
- /**
355
- * Stable identifier of the scenario being executed.
356
- *
357
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
358
- * input WITHOUT this field, substituting a sensible default
359
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
360
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
361
- * keeps the persisted shape unambiguous for downstream filters + aggregations
362
- * while removing the boilerplate of inventing placeholder ids at the call site.
363
- */
364
- scenarioId: string;
365
- variantId?: string;
366
- datasetVersion?: string;
367
- /** Git SHA of agent code at run time. */
368
- codeSha?: string;
369
- /** Hash of the prompt template + any system prompt. */
370
- promptSha?: string;
371
- /** Model id + date + system-prompt hash, concatenated. */
372
- modelFingerprint?: string;
373
- seed?: number;
374
- /** Arbitrary environment markers (shell, docker version, tz). */
375
- envFingerprint?: Record<string, string>;
376
- /** Version of the redaction rules applied to this run. */
377
- redactionVersion?: string;
378
- /** Parent run in a nested build workflow. A builder run's children are
379
- * app-build runs; those children are app-runtime runs. */
380
- parentRunId?: string;
381
- /** Stable project identifier — groups runs across chats + sessions. */
382
- projectId?: string;
383
- /** Chat/conversation identifier within a project. */
384
- chatId?: string;
385
- /** Layer classification — hint for aggregation; not enforced. */
386
- layer?: RunLayer;
387
- startedAt: number;
388
- endedAt?: number;
389
- status: RunStatus;
390
- outcome?: RunOutcome$1;
391
- budget?: BudgetSpec;
392
- /** Free-form labels for downstream grouping. */
393
- tags?: Record<string, string>;
394
- }
395
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
396
- type SpanStatus = 'ok' | 'error';
397
- interface SpanBase {
398
- spanId: string;
399
- parentSpanId?: string;
400
- runId: string;
401
- kind: SpanKind;
402
- name: string;
403
- startedAt: number;
404
- endedAt?: number;
405
- status?: SpanStatus;
406
- error?: string;
407
- /** Anything not covered by typed fields. Kept deliberately free-form. */
408
- attributes?: Record<string, unknown>;
409
- }
410
- interface Message {
411
- role: 'system' | 'user' | 'assistant' | 'tool';
412
- content: string;
413
- tokens?: number;
414
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
415
- images?: Array<{
416
- artifactId?: string;
417
- url?: string;
418
- mime?: string;
419
- }>;
420
- }
421
- interface LlmSpan extends SpanBase {
422
- kind: 'llm';
423
- model: string;
424
- messages: Message[];
425
- output?: string;
426
- inputTokens?: number;
427
- /** All generated tokens, including the reasoning subset when present. */
428
- outputTokens?: number;
429
- cachedTokens?: number;
430
- cacheWriteTokens?: number;
431
- /** Reasoning-token subset of `outputTokens`. */
432
- reasoningTokens?: number;
433
- costUsd?: number;
434
- finishReason?: string;
435
- }
436
- interface ToolSpan extends SpanBase {
437
- kind: 'tool';
438
- toolName: string;
439
- args: unknown;
440
- /** False when the source observed the call but did not capture its arguments. */
441
- argsCaptured?: boolean;
442
- result?: unknown;
443
- latencyMs?: number;
444
- }
445
- interface RetrievalSpan extends SpanBase {
446
- kind: 'retrieval';
447
- query: string;
448
- hits: Array<{
449
- docId: string;
450
- score: number;
451
- content?: string;
452
- }>;
453
- }
454
- interface JudgeSpan extends SpanBase {
455
- kind: 'judge';
456
- judgeId: string;
457
- /** Span this judgment applies to. */
458
- targetSpanId: string;
459
- dimension: string;
460
- /** Numeric score (free-range; interpretation up to the judge). */
461
- score: number;
462
- rationale?: string;
463
- evidence?: string;
464
- }
465
- interface SandboxSpan extends SpanBase {
466
- kind: 'sandbox';
467
- image?: string;
468
- command?: string;
469
- exitCode?: number;
470
- testsTotal?: number;
471
- testsPassed?: number;
472
- stdoutHash?: string;
473
- stderrHash?: string;
474
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
475
- wallMs?: number;
476
- }
477
- interface GenericSpan extends SpanBase {
478
- kind: 'agent' | 'custom';
479
- }
480
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
481
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
482
- interface TraceEvent {
483
- eventId: string;
484
- runId: string;
485
- spanId?: string;
486
- kind: EventKind;
487
- timestamp: number;
488
- payload: Record<string, unknown>;
489
- }
490
- interface BudgetLedgerEntry {
491
- runId: string;
492
- dimension: keyof BudgetSpec;
493
- limit: number;
494
- consumed: number;
495
- remaining: number;
496
- timestamp: number;
497
- breached: boolean;
498
- /** Span that triggered this entry, if any. */
499
- spanId?: string;
500
- }
501
- interface Artifact {
502
- artifactId: string;
503
- runId: string;
504
- spanId?: string;
505
- contentType: string;
506
- sizeBytes: number;
507
- /** sha256 in hex. */
508
- hash: string;
509
- /** External storage URL (R2, S3, filesystem path). */
510
- storageUrl?: string;
511
- /** Inline content for small blobs — keep under ~64KB. */
512
- inlineContent?: string;
513
- }
514
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
515
-
516
- /**
517
- * Paper-grade RunRecord schema + runtime validator.
518
- *
519
- * Every run that participates in a promotion gate, paper table, or
520
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
521
- * fields are exactly those the paper "Two Loops, Three Roles" requires
522
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
523
- * holdout split tag. A task score is optional because execution-only records
524
- * must preserve missing labels instead of converting errors into zero quality.
525
- *
526
- * This is intentionally NOT a replacement for the rich `Run` /
527
- * `ProposeReviewReport` / `ScenarioResult` types already in the
528
- * package. Those are runtime structures with full provenance. A
529
- * `RunRecord` is the analysis-time projection — the JSON-friendly
530
- * row you'd put in a parquet file or paste into a notebook.
531
- *
532
- * Validate at the boundary:
533
- *
534
- * const rec = validateRunRecord(rawJson) // throws on missing
535
- * const ok = isRunRecord(rawJson) // boolean check
536
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
537
- *
538
- * The validator runs in pure TS — zod is intentionally NOT a
539
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
540
- */
541
-
542
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
543
- * combined train+test pool that the optimizer is allowed to read. */
544
- type RunSplitTag = 'search' | 'dev' | 'holdout';
545
- /**
546
- * Explicit execution-lifecycle result for a run.
547
- *
548
- * This is separate from task quality (`outcome`) and failure classification.
549
- * Producers set it only from root-run or process evidence.
550
- */
551
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
552
- interface RunTokenUsage {
553
- input: number;
554
- /** All generated tokens charged as output, including reasoning tokens. */
555
- output: number;
556
- /** Reasoning-token subset of `output`, when the provider reports it. */
557
- reasoning?: number;
558
- /** Prompt tokens served from a provider cache. */
559
- cached?: number;
560
- /** Prompt tokens written into a provider cache. */
561
- cacheWrite?: number;
562
- }
563
- /**
564
- * How a run's USD amount was obtained.
565
- */
566
- type RunCostProvenance = {
567
- kind: 'observed';
568
- usd: number;
569
- } | {
570
- kind: 'estimated';
571
- usd: number;
572
- } | {
573
- kind: 'uncaptured';
574
- usd: null;
575
- };
576
- interface RunJudgeMetadata {
577
- model: string;
578
- promptVersion: string;
579
- /** [0,1] confidence the judge declared. Constant judge confidence
580
- * across many runs is a fallback signal (see `canary.ts`). */
581
- confidence: number;
582
- /** True if the judge degraded to a fallback path (rules-only,
583
- * prior-call cache, etc.). The canary uses this to alert. */
584
- fallback: boolean;
585
- }
586
- /**
587
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
588
- * judges over a multi-dimensional rubric.
589
- *
590
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
591
- * composite the gate uses. The full breakdown belongs here so consumers
592
- * can answer "which judge disagreed?", "which dimension dragged the
593
- * composite down?", and "did half the panel fail?" without re-running.
594
- *
595
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
596
- * `composite` are convenience projections — derivable but precomputed so
597
- * downstream IRR primitives (`interRaterReliability`,
598
- * `corpusInterRaterAgreement`) and reporters don't pay the same
599
- * aggregation twice.
600
- *
601
- * Fail-loud discipline: judges that errored out land in `failedJudges`
602
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
603
- * run); the explicit list makes a partial-failure recorded as such.
604
- */
605
- interface JudgeScoresRecord {
606
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
607
- perJudge: Record<string, Record<string, number>>;
608
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
609
- perDimMean: Record<string, number>;
610
- /** Composite mean across successful judges. Mirrors the task score only
611
- * when `failedJudges` is empty. */
612
- composite: number;
613
- /** Judges that errored or returned an unparseable verdict. Recorded
614
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
615
- * not inferred from missing keys in `perJudge`. */
616
- failedJudges?: string[];
617
- /** Free-form notes the judges emitted (joined across judges or
618
- * first-judge only — consumer's choice). */
619
- notes?: string;
620
- }
621
- interface RunOutcome {
622
- /** Score on the search/optimization split. Optional for holdout-only and
623
- * execution-only records. */
624
- searchScore?: number;
625
- /** Score on the held-out split. Optional for search-only and execution-only
626
- * records. When both scores are absent, the run is explicitly unlabeled. */
627
- holdoutScore?: number;
628
- /** Bag of any other metric the run produced — judge dimensions,
629
- * pass/fail counters, latency stats, etc. Numeric only — keeps
630
- * reporters honest. */
631
- raw: Record<string, number>;
632
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
633
- * judgements populate this; substrate primitives like
634
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
635
- * these records as input. Optional — single-judge or scalar-only
636
- * runs leave it unset. */
637
- judgeScores?: JudgeScoresRecord;
638
- /** Authenticity / realness verdict — did the run build the REAL thing on the
639
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
640
- * with an authenticity config populate it. Carried in the corpus so the
641
- * flywheel / off-policy learning can optimize for real completion, not gamed
642
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
643
- * must not count as a real success regardless of `score`. */
644
- realness?: {
645
- score: number;
646
- gated: boolean;
647
- reason?: string;
648
- };
649
- }
650
- /**
651
- * Mandatory paper-grade fields for a single evaluation run. Optional
652
- * fields are extension points; mandatory fields throw if missing.
653
- *
654
- * Hash discipline:
655
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
656
- * model (after any steering bundle merge).
657
- * - `configHash` is the sha256 of the effective run config (model,
658
- * temperature, tools, judges, splits). The pair (promptHash,
659
- * configHash) uniquely identifies an experiment cell.
660
- *
661
- * Model snapshot discipline:
662
- * - `model` MUST encode a snapshot version. Bare aliases like
663
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
664
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
665
- */
666
- interface RunRecord {
667
- /** UUID for the run. */
668
- runId: string;
669
- /** Logical experiment grouping (a treatment vs a baseline within
670
- * the same sweep should share `experimentId`). */
671
- experimentId: string;
672
- /** Stable identifier for the candidate (variant) being run. The
673
- * promotion gate compares two `candidateId`s on matched items. */
674
- candidateId: string;
675
- /** RNG seed for the run. Always recorded — silent re-seeding is
676
- * the most common cause of non-reproducible numbers. */
677
- seed: number;
678
- /** Model identifier WITH snapshot version. */
679
- model: string;
680
- /** sha256 of the effective prompt (post-steering). */
681
- promptHash: string;
682
- /** sha256 of the effective config. */
683
- configHash: string;
684
- /** Git SHA the harness was run from. */
685
- commitSha: string;
686
- /** End-to-end wall-clock duration in milliseconds. */
687
- wallMs: number;
688
- /** Time spent queued before execution started, if known. */
689
- queueMs?: number;
690
- /** Total USD cost, or null when the producer could not capture one. */
691
- costUsd: number | null;
692
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
693
- costProvenance: RunCostProvenance;
694
- /** Token usage breakdown. */
695
- tokenUsage: RunTokenUsage;
696
- /** Root-run or process terminal result. Never inferred from a child span. */
697
- terminalOutcome: RunTerminalOutcome;
698
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
699
- * or incomplete terminal result; never populated from a child span. */
700
- terminalFailureReason?: string;
701
- /** Judge-side metadata, if a judge was used. */
702
- judgeMetadata?: RunJudgeMetadata;
703
- /** Per-split scores + raw bag. */
704
- outcome: RunOutcome;
705
- /** Canonical task-failure class drawn from the shared
706
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
707
- * evidence. Execution errors belong in
708
- * `outcome.raw.execution_error_count`. */
709
- failureClass?: FailureClass;
710
- /** Free-form task-failure detail scoped under a non-success
711
- * `failureClass`. It is invalid without that class. */
712
- failureMode?: string;
713
- /** Which split this run was drawn from. */
714
- splitTag: RunSplitTag;
715
- /**
716
- * Stable scenario identifier the run observed or was scored against.
717
- * Comparison primitives match this identity rather than input order.
718
- */
719
- scenarioId: string;
720
- /**
721
- * Canonical identity for the agent profile cell that produced this row:
722
- * profile artifact hash plus optional harness/model/prompt/reporting
723
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
724
- * longitudinal reports by the complete source profile, not by a loose
725
- * candidate label or opaque config hash.
726
- */
727
- agentProfile?: AgentProfileCell;
728
- }
729
-
730
- interface RunFilter {
731
- scenarioId?: string;
732
- variantId?: string;
733
- status?: RunStatus;
734
- since?: number;
735
- until?: number;
736
- tag?: {
737
- key: string;
738
- value: string;
739
- };
740
- parentRunId?: string;
741
- projectId?: string;
742
- chatId?: string;
743
- layer?: RunLayer;
744
- }
745
- interface SpanFilter {
746
- runId?: string;
747
- parentSpanId?: string;
748
- kind?: SpanKind;
749
- name?: string;
750
- toolName?: string;
751
- judgeId?: string;
752
- since?: number;
753
- until?: number;
754
- }
755
- interface EventFilter {
756
- runId?: string;
757
- spanId?: string;
758
- kind?: EventKind;
759
- since?: number;
760
- until?: number;
761
- }
762
- interface TraceStore {
763
- appendRun(run: Run): Promise<void>;
764
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
765
- appendSpan(span: Span): Promise<void>;
766
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
767
- appendEvent(event: TraceEvent): Promise<void>;
768
- appendArtifact(artifact: Artifact): Promise<void>;
769
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
770
- getRun(runId: string): Promise<Run | undefined>;
771
- listRuns(filter?: RunFilter): Promise<Run[]>;
772
- spans(filter?: SpanFilter): Promise<Span[]>;
773
- events(filter?: EventFilter): Promise<TraceEvent[]>;
774
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
775
- artifacts(runId: string): Promise<Artifact[]>;
776
- }
777
-
778
- /**
779
- * Rollout minting — `tangle.rollout.v1` lines joined from the records the
780
- * substrate ALREADY keeps. There is no separate rollout store: a rollout
781
- * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with
782
- * its trace (spans share `runId`), projected into the canonical line.
783
- *
784
- * Composition, not duplication:
785
- * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)
786
- * - step structure → `buildTrajectory` over the shared TraceStore
787
- * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
788
- * - PRM / reward-model → `reward-model-export.ts`
789
- *
790
- * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true
791
- * is never exported with a positive reward — the gate travels into the
792
- * training data (`reward` forced to 0, `realness_gated: true`), so a
793
- * fine-tune cannot learn from gamed successes.
794
- *
795
- * Records without spans become labeled GAP LINES (messages: [],
796
- * provenance.gap) — present in the output AND surfaced in
797
- * `missingTraces`; a capture gap is a finding, never a silent omission.
798
- */
799
-
800
- /** Redactor applied to every exported string (secrets, PII). Identity by default. */
801
- type RolloutScrubber = (text: string) => string;
802
- interface MintRolloutOptions {
803
- scrub?: RolloutScrubber;
804
- /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */
805
- maxSteps?: number;
806
- /** Role recorded on every minted line. Default 'agent' (a solo eval run). */
807
- role?: RolloutRole;
808
- /** Task suite label. Default: the record's `experimentId`. */
809
- suite?: string;
810
- /** Injected clock for deterministic output. */
811
- now?: () => Date;
812
- }
813
- interface MintRolloutResult {
814
- rows: RolloutLine[];
815
- /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
816
- missingTraces: string[];
817
- }
818
- declare function rolloutReward(record: RunRecord): {
819
- reward: number;
820
- gated: boolean;
821
- };
822
- /**
823
- * Join RunRecords with their traces into canonical rollout lines. Records
824
- * without spans are emitted as labeled gap lines and reported in
825
- * `missingTraces`. Execution-only records without a task score are rejected
826
- * because a missing training label is not a zero reward.
827
- */
828
- declare function mintRolloutRows(records: RunRecord[], store: TraceStore, options?: MintRolloutOptions): Promise<MintRolloutResult>;
829
-
830
- /**
831
- * Backfill reader over Claude Code project transcripts
832
- * (~/.claude/projects/<cwd-slug>/<sessionId>.jsonl) → canonical
833
- * chat-with-tools messages plus per-session token usage.
834
- *
835
- * Transcript lines consumed: type:"user" (string content or content blocks —
836
- * text + tool_result) and type:"assistant" (content blocks — thinking, text,
837
- * tool_use; message.usage carries tokens). Sidechain lines (isSidechain=true,
838
- * subagent threads) are separate invocations and are excluded from the main
839
- * transcript. Everything else (queue-operation, attachment, last-prompt…) is
840
- * transport metadata, not conversation.
841
- */
842
-
843
- declare const DEFAULT_CLAUDE_PROJECTS_DIR: string;
844
- /** Claude Code's project-directory slug for a working directory. */
845
- declare function claudeProjectSlug(cwd: string): string;
846
- interface ClaudeTranscriptRef {
847
- sessionId: string;
848
- path: string;
849
- }
850
- /** Transcript files recorded for sessions launched from `cwd`. */
851
- declare function findClaudeTranscripts(cwd: string, projectsDir?: string): Promise<ClaudeTranscriptRef[]>;
852
- interface ClaudeUsageTotals {
853
- tokensIn: number;
854
- tokensOut: number;
855
- cacheRead: number;
856
- cacheWrite: number;
857
- }
858
- interface ClaudeTranscript {
859
- messages: ChatMessage[];
860
- usage: ClaudeUsageTotals;
861
- /** Timestamp of the first conversation line; null = empty transcript. */
862
- startedAt: string | null;
863
- endedAt: string | null;
864
- model: string | null;
865
- }
866
- interface ReadClaudeTranscriptOptions {
867
- /**
868
- * Read the sidechain (subagent) thread instead of skipping it. Subagent
869
- * transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain
870
- * lines end to end, so their usage is invisible without this.
871
- */
872
- readonly includeSidechain?: boolean;
873
- }
874
- /** Parse one transcript jsonl into canonical messages + usage totals. */
875
- declare function readClaudeTranscript(path: string, options?: ReadClaudeTranscriptOptions): Promise<ClaudeTranscript>;
876
-
877
- /**
878
- * Read-only backfill reader over the opencode sqlite store
879
- * (~/.local/share/opencode/opencode.db) → canonical chat-with-tools messages.
880
- *
881
- * Schema consumed (observed, 2026-07): `session` rows carry directory /
882
- * parent_id / agent / model / cost / tokens_*; `message` rows carry a JSON
883
- * `data` blob ({role, modelID, providerID, tokens, cost, finish}); `part`
884
- * rows carry the actual content ({type: text|reasoning|tool|step-start|
885
- * step-finish|snapshot…}). Tool parts hold {callID, state:{input, output,
886
- * status}} — both the call and its result, which we split into an assistant
887
- * tool_call plus a role:"tool" result message.
888
- *
889
- * The store is mutable and can be corrupt (a `.corrupt-bak` sibling ships
890
- * next to it in the wild), so `openOpencodeDb` returns null instead of
891
- * throwing — callers record a gap line, never crash the backfill.
892
- */
893
-
894
- declare const DEFAULT_OPENCODE_DB: string;
895
- interface OpencodeSessionRow {
896
- id: string;
897
- parentId: string | null;
898
- directory: string;
899
- agent: string | null;
900
- /** Raw session.model JSON: {id, providerID, variant} where present. */
901
- model: {
902
- id?: string;
903
- providerID?: string;
904
- } | null;
905
- costUsd: number;
906
- tokensInput: number;
907
- tokensOutput: number;
908
- tokensReasoning: number;
909
- tokensCacheRead: number;
910
- tokensCacheWrite: number;
911
- timeCreated: number;
912
- timeUpdated: number;
913
- }
914
- /** Open the store read-only; null = unavailable/corrupt (caller records a gap). */
915
- declare function openOpencodeDb(path?: string): Promise<DatabaseSync | null>;
916
- /** Sessions whose cwd is `directory` (the worker-clone join key). */
917
- declare function findOpencodeSessionsByDirectory(db: DatabaseSync, directory: string): OpencodeSessionRow[];
918
- declare function findOpencodeSessionById(db: DatabaseSync, sessionId: string): OpencodeSessionRow | null;
919
- /**
920
- * Convert one session's message+part rows into canonical messages.
921
- * An opencode assistant message row spans several model steps; each step's
922
- * parts (reasoning → text → tool …) become one assistant message followed by
923
- * the role:"tool" results of its calls, preserving order.
924
- */
925
- declare function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[];
926
-
927
- /**
928
- * Deterministic scrubbing pass over rollout-ledger lines before public release.
929
- *
930
- * Every rule is a pure regex rewrite applied to every string value in a line
931
- * (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed
932
- * line is still a valid `tangle.rollout.v1` line. Rules are idempotent:
933
- * scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is
934
- * the property the release pipeline relies on to prove nothing half-scrubbed
935
- * ships. Rule order matters: whole `KEY=value` env pairs are redacted before
936
- * the bare-key rule so one secret is never counted twice.
937
- */
938
-
939
- interface ScrubRule {
940
- name: string;
941
- pattern: RegExp;
942
- /** Rewrite for one match; `g1` is the first capture group when present. */
943
- rewrite: (match: string, g1?: string) => string;
944
- }
945
- declare const SCRUB_RULES: readonly ScrubRule[];
946
- /** Rule name → number of matches rewritten. Always carries every rule (0 is data). */
947
- type ScrubCounts = Record<string, number>;
948
- declare function emptyScrubCounts(): ScrubCounts;
949
- declare function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts;
950
- declare function scrubText(text: string, counts: ScrubCounts): string;
951
- /** Scrub every string value in a line; structure and key order are preserved. */
952
- declare function scrubRolloutLine(line: RolloutLine, counts: ScrubCounts): RolloutLine;
953
- declare function scrubLines(lines: RolloutLine[]): {
954
- lines: RolloutLine[];
955
- counts: ScrubCounts;
956
- };
957
- /**
958
- * A `RolloutScrubber` (text → text) applying the full rule set — the
959
- * default hook to pass to `mintRolloutRows({ scrub })` so lines are
960
- * scrubbed at mint time, before they ever reach a ledger file. Release
961
- * builds re-run `scrubLines` regardless (idempotent), so double-scrubbing
962
- * is safe and counted as zero.
963
- */
964
- declare function defaultRolloutScrubber(text: string): string;
965
-
966
- /**
967
- * HuggingFace dataset-card (README.md) generation for a rollout-ledger release.
968
- *
969
- * The card is a pure function of the SCRUBBED lines plus the release options —
970
- * no timestamps, no environment reads — so rebuilding from the same ledger
971
- * yields byte-identical output. It documents the schema, provenance (run ids,
972
- * generations, the official judge), per-role reward semantics including the
973
- * inherited/contribution caveat, and a role × reward counts table.
974
- */
975
-
976
- declare const RELEASE_FORMATS: readonly ["sft", "verifiers", "rft", "raw"];
977
- type ReleaseFormat = (typeof RELEASE_FORMATS)[number];
978
- /** Format → data file path inside the dataset dir (train split only). */
979
- declare const FORMAT_FILES: Record<ReleaseFormat, string>;
980
- interface DatasetCardInputs {
981
- /** Scrubbed, release-filtered lines (what actually ships). */
982
- lines: RolloutLine[];
983
- formats: ReleaseFormat[];
984
- includeProposers: boolean;
985
- /** Source ledger basenames, for provenance. */
986
- sourceFiles: string[];
987
- scrubTotals: ScrubCounts;
988
- excluded: {
989
- proposers: number;
990
- nonTrain: number;
991
- };
992
- formatCounts: Partial<Record<ReleaseFormat, number>>;
993
- }
994
- declare function buildDatasetCard(inputs: DatasetCardInputs): string;
995
-
996
- /**
997
- * One-command HuggingFace dataset release from rollout ledgers:
998
- *
999
- * agent-eval rollout-release <ledger.jsonl...> --out <dir> \
1000
- * [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]
1001
- *
1002
- * Pipeline per input ledger: read + validate → fail-closed filters
1003
- * (trainable split only; proposer sessions dropped unless
1004
- * --include-proposers, they contain improvement-loop harness source) →
1005
- * deterministic scrub → export the requested formats + scrub-report.json +
1006
- * auto-generated README.md card. Deterministic: same inputs and flags →
1007
- * byte-identical output dir.
1008
- *
1009
- * --push uploads the built dir with `huggingface-cli upload` only when the
1010
- * CLI exists on PATH and HF_TOKEN is present in the env; the token is
1011
- * never printed. Everything else runs fully offline.
1012
- */
1013
-
1014
- interface BuildOptions {
1015
- out: string;
1016
- formats: ReleaseFormat[];
1017
- includeProposers: boolean;
1018
- }
1019
- interface ScrubReport {
1020
- /** Input ledger path → rule → rewrite count (only shipped lines are scrubbed). */
1021
- files: Record<string, ScrubCounts>;
1022
- totals: ScrubCounts;
1023
- excluded: {
1024
- proposers: number;
1025
- nonTrain: number;
1026
- };
1027
- }
1028
- interface BuildSummary {
1029
- inputs: string[];
1030
- read: number;
1031
- kept: number;
1032
- scrub: ScrubReport;
1033
- formatCounts: Partial<Record<ReleaseFormat, number>>;
1034
- files: string[];
1035
- }
1036
- declare function buildHfDataset(inputs: string[], options: BuildOptions): Promise<BuildSummary>;
1037
- declare function planPushCommand(repo: string, outDir: string): string[];
1038
- declare function pushDataset(repo: string, outDir: string): void;
1039
- interface RolloutReleaseCliArgs extends BuildOptions {
1040
- inputs: string[];
1041
- push: string | null;
1042
- }
1043
- declare const ROLLOUT_RELEASE_USAGE = "usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]";
1044
- declare function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs;
1045
- /** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
1046
- declare function runRolloutReleaseCli(argv: string[]): Promise<number>;
1047
-
1048
- export { type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, FORMAT_FILES, type MintRolloutOptions, type MintRolloutResult, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type ReleaseFormat, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, isRolloutLine, isTrainableSplit, mintRolloutRows, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutLedger, rolloutReward, runRolloutReleaseCli, scrubLines, scrubRolloutLine, scrubText, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, validateRolloutLine, writeRolloutLedger };
1
+ import { A as isTrainableSplit, C as TRAINABLE_SPLITS, D as assertRolloutLine, E as assertMintedLines, O as gateGamedOutcome, S as RolloutTask, T as assertMinted, _ as RolloutPolicy, a as GatedEvidence, b as RolloutSplit, c as ROLLOUT_CAPTURES, d as ROLLOUT_SPLITS, f as RolloutArtifacts, g as RolloutOutcome, h as RolloutLine, i as ChatToolCall, j as validateRolloutLine, k as isRolloutLine, l as ROLLOUT_ROLES, m as RolloutCostBlock, n as ChatMessage, o as MintedRolloutLine, p as RolloutCapture, r as ChatRole, s as MintedRolloutOutcome, t as CHAT_ROLES, u as ROLLOUT_SCHEMA, v as RolloutProvenance, w as ToolDef, x as RolloutStep, y as RolloutRole } from "../schema-Cef2cFmb.js";
2
+ import { $ as ScorePreference, $t as toVerifiersRolloutOutput, A as ReleaseRowRef, At as GATE_CHECK_IDS, B as readOpencodeSessionMessages, Bt as RealnessLabels, C as scrubRolloutLine, Ct as HarborToolCall, D as FormatGateCounts, Dt as toHarborTrajectories, E as FORMAT_GATE_DISPOSITION, Et as relabelImportedSplit, F as DEFAULT_OPENCODE_DB, Ft as GateCheckedOutcome, G as claudeProjectSlug, Gt as VerifiersRolloutOutput, H as ClaudeTranscriptRef, Ht as RftItem, I as OpencodeSessionRow, It as GateEntryPoint, J as MintRolloutOptions, Jt as toJsonl, K as findClaudeTranscripts, Kt as VerifiersTokenUsage, L as findOpencodeSessionById, Lt as GatePolicy, M as gatedRolloutIds, Mt as GateCheck, N as measureFormatGate, Nt as GateCheckDisposition, O as GateDisposition, Ot as toHarborTrajectory, P as releaseRowRefs, Pt as GateCheckId, Q as ScoreOrigin, Qt as toSftRows, R as findOpencodeSessionsByDirectory, Rt as gateErrors, S as scrubLines, St as HarborSubagentTrajectoryRef, T as EmittedEvidence, Tt as fromHarborTrajectory, U as ClaudeUsageTotals, Ut as SftExportOptions, V as ClaudeTranscript, Vt as RewardRow, W as DEFAULT_CLAUDE_PROJECTS_DIR, Wt as SftRow, X as RolloutScrubber, Xt as toRftItem, Y as MintRolloutResult, Yt as toRewardRows, Z as mintRolloutRows, Zt as toRftItems, _ as ScrubCounts, _t as HarborMetrics, a as ScrubReport, at as trainingScore, b as defaultRolloutScrubber, bt as HarborStep, c as planPushCommand, ct as readRolloutLedger, d as DatasetCardInputs, dt as FromHarborOptions, en as toVerifiersRolloutOutputs, et as isRealnessGated, f as FORMAT_FILES, ft as HARBOR_IMPORT_GAP, g as SCRUB_RULES, gt as HarborImageSource, h as buildDatasetCard, ht as HarborFinalMetrics, i as RolloutReleaseCliArgs, it as trainingReward, j as assertGateReport, jt as GATE_POLICIES, k as GateReport, kt as GATE_CHECKS, l as pushDataset, lt as writeRolloutLedger, m as ReleaseFormat, mt as HarborContentPart, n as BuildSummary, nt as observedSplitScore, o as buildHfDataset, ot as appendRolloutLines, p as RELEASE_FORMATS, pt as HarborAgent, q as readClaudeTranscript, qt as realnessLabels, r as ROLLOUT_RELEASE_USAGE, rt as scoreOrigin, s as parseRolloutReleaseArgs, st as readRolloutJournal, t as BuildOptions, tt as observedScore, u as runRolloutReleaseCli, ut as ATIF_SCHEMA_VERSION, v as ScrubRule, vt as HarborObservation, w as scrubText, wt as HarborTrajectory, x as emptyScrubCounts, xt as HarborStepSource, y as addScrubCounts, yt as HarborObservationResult, z as openOpencodeDb, zt as gatedEvidenceOf } from "../index-2JJSA6-r2.js";
3
+ export { ATIF_SCHEMA_VERSION, type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, type EmittedEvidence, FORMAT_FILES, FORMAT_GATE_DISPOSITION, type FormatGateCounts, type FromHarborOptions, GATE_CHECKS, GATE_CHECK_IDS, GATE_POLICIES, type GateCheck, type GateCheckDisposition, type GateCheckId, type GateCheckedOutcome, type GateDisposition, type GateEntryPoint, type GatePolicy, type GateReport, type GatedEvidence, HARBOR_IMPORT_GAP, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type RealnessLabels, type ReleaseFormat, type ReleaseRowRef, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScoreOrigin, type ScorePreference, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertGateReport, assertMinted, assertMintedLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, fromHarborTrajectory, gateErrors, gateGamedOutcome, gatedEvidenceOf, gatedRolloutIds, isRealnessGated, isRolloutLine, isTrainableSplit, measureFormatGate, mintRolloutRows, observedScore, observedSplitScore, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutJournal, readRolloutLedger, realnessLabels, relabelImportedSplit, releaseRowRefs, runRolloutReleaseCli, scoreOrigin, scrubLines, scrubRolloutLine, scrubText, toHarborTrajectories, toHarborTrajectory, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, trainingReward, trainingScore, validateRolloutLine, writeRolloutLedger };