@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,763 @@
1
+ import { _ as GATE_POLICIES, b as undeclaredStepPayload, g as GATE_CHECK_IDS, i as ROLLOUT_SCHEMA, p as isTrainableSplit, r as ROLLOUT_ROLES, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
2
+ import { l as toVerifiersRolloutOutputs, o as toRftItems, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
3
+ import { basename, dirname, join } from "node:path";
4
+ import { appendFile, mkdir, readFile, writeFile } from "node:fs/promises";
5
+ import { spawnSync } from "node:child_process";
6
+ //#region src/rollout/ledger.ts
7
+ /**
8
+ * Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`
9
+ * lines. Writes validate BEFORE touching disk (a bad line never lands);
10
+ * reads validate line-by-line and fail loud with the line number, because a
11
+ * silently-skipped rollout is a corrupted dataset.
12
+ *
13
+ * "Validate" includes the anti-Goodhart invariant (a realness-gated line may
14
+ * not carry a positive reward), so a poisoned line can neither enter a ledger
15
+ * nor leave one.
16
+ *
17
+ * Two read modes, matching the two write-side row classes: `readRolloutLedger`
18
+ * re-validates under the mint policy (training data), `readRolloutJournal`
19
+ * under the write policy (supervision journals, whose unscreened positive
20
+ * rewards are writable and must stay readable).
21
+ */
22
+ function serialize(lines) {
23
+ for (const [i, line] of lines.entries()) assertRolloutLine(line, `rollout line [${i}]`);
24
+ return lines.map((line) => JSON.stringify(line)).join("\n") + (lines.length > 0 ? "\n" : "");
25
+ }
26
+ /** Replace the ledger file with exactly `lines`. */
27
+ async function writeRolloutLedger(path, lines) {
28
+ const payload = serialize(lines);
29
+ await mkdir(dirname(path), { recursive: true });
30
+ await writeFile(path, payload);
31
+ }
32
+ /** Append `lines` to the ledger file (created if absent). */
33
+ async function appendRolloutLines(path, lines) {
34
+ if (lines.length === 0) return;
35
+ const payload = serialize(lines);
36
+ await mkdir(dirname(path), { recursive: true });
37
+ await appendFile(path, payload);
38
+ }
39
+ /**
40
+ * Read and validate every line. Throws on the first malformed/invalid line
41
+ * (with its 1-based line number) — fail-closed, never a silent drop.
42
+ *
43
+ * Validation includes the anti-Goodhart invariant, which is why the result is
44
+ * `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this
45
+ * process from outside the type system (another run, another machine, a
46
+ * hand-edited JSONL), so this read is the runtime boundary where a poisoned
47
+ * line is refused rather than exported.
48
+ */
49
+ async function readRolloutLedger(path) {
50
+ return readLines(path, (parsed, context) => assertMinted(parsed, context));
51
+ }
52
+ /**
53
+ * Read a ledger under the WRITE-side policy (`validateRolloutLine`), which
54
+ * omits the unscreened-reward check. `writeRolloutLedger` accepts a
55
+ * supervision-journal row (`realness_screened: false` with a positive reward
56
+ * — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says
57
+ * such rows "must stay writable, readable and reportable"; a read API that
58
+ * only re-validated under `assertMinted` made every such file unreadable —
59
+ * write-accepted but read-refused is a data-loss trap.
60
+ *
61
+ * The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here
62
+ * can reach a training exporter without passing `assertMinted`, so the
63
+ * promotion gate (which DOES enforce unscreened-reward) is exactly as closed
64
+ * as before. Use `readRolloutLedger` when the file is training data.
65
+ */
66
+ async function readRolloutJournal(path) {
67
+ return readLines(path, (parsed, context) => {
68
+ assertRolloutLine(parsed, context);
69
+ return parsed;
70
+ });
71
+ }
72
+ async function readLines(path, admit) {
73
+ const raw = await readFile(path, "utf8");
74
+ const lines = [];
75
+ const rawLines = raw.split("\n");
76
+ for (let i = 0; i < rawLines.length; i++) {
77
+ const text = rawLines[i];
78
+ if (!text?.trim()) continue;
79
+ let parsed;
80
+ try {
81
+ parsed = JSON.parse(text);
82
+ } catch (error) {
83
+ throw new Error(`${path}:${i + 1}: malformed JSON — ${error instanceof Error ? error.message : String(error)}`);
84
+ }
85
+ lines.push(admit(parsed, `${path}:${i + 1}`));
86
+ }
87
+ return lines;
88
+ }
89
+ //#endregion
90
+ //#region src/rollout/release/gate-report.ts
91
+ const FORMAT_GATE_DISPOSITION = {
92
+ sft: "exclude",
93
+ verifiers: "zero-and-flag",
94
+ rft: "zero-and-flag",
95
+ raw: "zero-and-flag"
96
+ };
97
+ /** Rollout ids of every gated line, the key the emitted rows are matched on. */
98
+ function gatedRolloutIds(lines) {
99
+ return new Set(lines.filter((line) => line.outcome.realness_gated).map((line) => line.rollout_id));
100
+ }
101
+ /**
102
+ * Row refs per format. Written as one adapter per format so that the knowledge
103
+ * of WHERE the id and reward live in each published shape sits next to the
104
+ * assertion that uses it — an exporter that moves either field breaks here
105
+ * rather than silently reporting zero gated rows.
106
+ */
107
+ const releaseRowRefs = {
108
+ sft: (rows) => rows.map((row) => ({
109
+ rollout_id: row.metadata.rollout_id,
110
+ reward: row.metadata.reward,
111
+ realness_screened: row.metadata.realness_screened
112
+ })),
113
+ verifiers: (rows) => rows.map((row) => ({
114
+ rollout_id: row.info.rollout_id,
115
+ reward: row.reward,
116
+ evidence: row.metrics,
117
+ realness_screened: row.info.realness_screened
118
+ })),
119
+ rft: (rows) => rows.map((row) => ({
120
+ rollout_id: row.reference.rollout_id,
121
+ reward: row.reference.reward,
122
+ evidence: row.reference.verdict,
123
+ realness_screened: row.reference.realness_screened
124
+ })),
125
+ raw: (lines) => lines.map((line) => ({
126
+ rollout_id: line.rollout_id,
127
+ reward: line.outcome.reward,
128
+ evidence: {
129
+ metrics: line.outcome.metrics,
130
+ verdict: line.outcome.verdict
131
+ },
132
+ stepEvidence: undeclaredStepPayload(line.steps),
133
+ realness_screened: line.outcome.realness_screened ?? null
134
+ }))
135
+ };
136
+ /** Every positive finite number inside a row's reward-derived payload, with its path. */
137
+ function positiveNumbersIn(value, path) {
138
+ if (typeof value === "number") return Number.isFinite(value) && value > 0 ? [{
139
+ path,
140
+ value
141
+ }] : [];
142
+ if (Array.isArray(value)) return value.flatMap((item, i) => positiveNumbersIn(item, `${path}[${i}]`));
143
+ if (typeof value === "object" && value !== null) return Object.entries(value).flatMap(([key, child]) => positiveNumbersIn(child, path === "" ? key : `${path}.${key}`));
144
+ return [];
145
+ }
146
+ /** Measure one format's gated rows from the refs of the rows about to be written. */
147
+ function measureFormatGate(gated, refs) {
148
+ const emitted = refs.filter((ref) => gated.has(ref.rollout_id));
149
+ const rewards = emitted.map((ref) => ref.reward).filter((r) => r !== null);
150
+ const evidence = emitted.flatMap((ref) => positiveNumbersIn(ref.evidence, ""));
151
+ const stepEvidence = emitted.flatMap((ref) => positiveNumbersIn(ref.stepEvidence, "steps"));
152
+ const unscreened = refs.filter((ref) => ref.realness_screened === false).map((ref) => ref.reward).filter((r) => r !== null && r > 0);
153
+ return {
154
+ input: gated.size,
155
+ emitted: emitted.length,
156
+ excluded: gated.size - emitted.length,
157
+ maxEmittedReward: rewards.length === 0 ? null : Math.max(...rewards),
158
+ maxEmittedEvidence: evidence.reduce((best, found) => best === null || found.value > best.value ? found : best, null),
159
+ unscreenedPositiveRows: unscreened.length,
160
+ maxUnscreenedReward: unscreened.length === 0 ? null : Math.max(...unscreened),
161
+ maxEmittedStepEvidence: stepEvidence.reduce((best, found) => best === null || found.value > best.value ? found : best, null)
162
+ };
163
+ }
164
+ /**
165
+ * The measured form of each canonical gate check, over the rows a release is
166
+ * ABOUT TO WRITE. Returns the failure message, or `null` when the format is
167
+ * clean on that check.
168
+ *
169
+ * TOTAL over `GateCheckId` — this map and `GATE_POLICIES.assertGateReport` are
170
+ * the two things a new check breaks here, so the release certifier cannot be
171
+ * left behind by a check added anywhere else in the package. That is the whole
172
+ * point: for four rounds each guard composed its own subset by hand, and a
173
+ * release certifying CLEAN while leaking is the most expensive version of that
174
+ * mistake, because it is the leak plus a document saying there isn't one.
175
+ */
176
+ const REPORT_MEASURES = {
177
+ "reward-relationship": (format, counts) => counts.maxEmittedReward === null || counts.maxEmittedReward <= 0 ? null : `release format "${format}": ${counts.emitted} realness-gated row(s) carry a positive reward (max ${counts.maxEmittedReward}). A run flagged as gamed may not ship a positive reward in any config.`,
178
+ "gated-evidence": (format, counts) => {
179
+ if (counts.maxEmittedEvidence === null) return null;
180
+ const { path, value } = counts.maxEmittedEvidence;
181
+ return `release format "${format}": ${counts.emitted} realness-gated row(s) carry a positive reward-derived number (${path} = ${value}). Zeroing the scalar is not enough — the per-layer scores and judge verdict a fabricated reward was computed FROM are the same signal in component form, and in this format they are read as training input. They belong in \`provenance.gated_evidence\` (see \`gateGamedOutcome\`), not on the row.`;
182
+ },
183
+ "undeclared-step-payload": (format, counts) => {
184
+ if (counts.maxEmittedStepEvidence === null) return null;
185
+ const { path, value } = counts.maxEmittedStepEvidence;
186
+ return `release format "${format}": ${counts.emitted} realness-gated row(s) carry a positive per-step number under a field \`tangle.rollout.v1\` does not declare (${path} = ${value}). A per-step reward is the same training signal as the scalar, in credit-assignment form, and the exporters copy \`steps\` through verbatim — so the row ships it beside a \`reward\` of 0. It belongs in \`provenance.gated_evidence.steps\` (see \`gateGamedOutcome\`).`;
187
+ },
188
+ "unscreened-reward": (format, counts) => counts.maxUnscreenedReward === null ? null : `release format "${format}": ${counts.unscreenedPositiveRows} row(s) carry a positive reward (max ${counts.maxUnscreenedReward}) whose producer declares that NO authenticity screen ran on it (\`realness_screened: false\`). Nothing has established those successes are real, and a published dataset may not present an unqualified verdict as a measured one. Screen the runs, or publish them at \`reward: null\`.`
189
+ };
190
+ /**
191
+ * Fail the build when the measurement disagrees with the declared policy.
192
+ *
193
+ * Throws, never filters: an emitted positive reward on a gated row means an
194
+ * exporter upstream stopped applying the gate, and silently dropping the row
195
+ * would hide the producer that made it — the producer is the actual defect.
196
+ *
197
+ * Certifies the whole emitted outcome, not `reward` alone. The earlier version
198
+ * checked one field and therefore certified a release CLEAN while its
199
+ * `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in
200
+ * the top-level `metrics` dict — the card then rendered "max reward | 0" over
201
+ * exactly that file. A certification that is wrong is worse than an
202
+ * uncertified leak, so the checks it runs are no longer written down here at
203
+ * all: it iterates `GATE_CHECK_IDS` under its own declared policy.
204
+ */
205
+ function assertGateReport(report) {
206
+ for (const [format, counts] of Object.entries(report.byFormat)) {
207
+ for (const id of GATE_CHECK_IDS) {
208
+ if (GATE_POLICIES.assertGateReport[id].kind !== "enforce") continue;
209
+ const failure = REPORT_MEASURES[id](format, counts);
210
+ if (failure !== null) throw new Error(failure);
211
+ }
212
+ if (FORMAT_GATE_DISPOSITION[format] === "exclude" && counts.emitted > 0) throw new Error(`release format "${format}" declares gated lines EXCLUDED but wrote ${counts.emitted} of them.`);
213
+ }
214
+ }
215
+ //#endregion
216
+ //#region src/rollout/release/card.ts
217
+ /**
218
+ * HuggingFace dataset-card (README.md) generation for a rollout-ledger release.
219
+ *
220
+ * The card is a pure function of the SCRUBBED lines plus the release options —
221
+ * no timestamps, no environment reads — so rebuilding from the same ledger
222
+ * yields byte-identical output. It documents the schema, provenance (run ids,
223
+ * generations, the official judge), per-role reward semantics including the
224
+ * inherited/contribution caveat, and a role × reward counts table.
225
+ */
226
+ const RELEASE_FORMATS = [
227
+ "sft",
228
+ "verifiers",
229
+ "rft",
230
+ "raw"
231
+ ];
232
+ /** Format → data file path inside the dataset dir (train split only). */
233
+ const FORMAT_FILES = {
234
+ sft: "sft/train.jsonl",
235
+ verifiers: "verifiers/train.jsonl",
236
+ rft: "rft/train.jsonl",
237
+ raw: "raw/train.jsonl"
238
+ };
239
+ const FORMAT_DESCRIPTIONS = {
240
+ sft: "Successful trainable-split transcripts (`reward >= 1`, never realness-gated) as `{messages, metadata}` chat JSONL.",
241
+ verifiers: "Prime Intellect verifiers `RolloutOutput`: prompt/completion split at the first assistant turn, plus reward, metrics, tool defs, and token usage.",
242
+ rft: "OpenAI RFT items: prompt turns plus `reference.*` verdict fields for a grader (completions are re-sampled during RFT).",
243
+ raw: `Full \`${ROLLOUT_SCHEMA}\` ledger lines (scrubbed), one per agent invocation.`
244
+ };
245
+ function unique(values) {
246
+ return [...new Set(values.filter((v) => v !== null))].sort();
247
+ }
248
+ function formatReward(reward) {
249
+ if (reward === null) return "null";
250
+ return Number.isInteger(reward) ? String(reward) : reward.toFixed(4);
251
+ }
252
+ function markdownTable(header, rows) {
253
+ return [
254
+ `| ${header.join(" | ")} |`,
255
+ `| ${header.map(() => "---").join(" | ")} |`,
256
+ ...rows.map((row) => `| ${row.join(" | ")} |`)
257
+ ].join("\n");
258
+ }
259
+ /** Where the anti-Goodhart flag lives on each format's emitted row. */
260
+ const GATE_FLAG_FIELD = {
261
+ sft: "— (no gated row is written)",
262
+ verifiers: "`info.realness_gated`",
263
+ rft: "`reference.realness_gated`",
264
+ raw: "`outcome.realness_gated`"
265
+ };
266
+ const GATE_POLICY_LABEL = {
267
+ sft: "EXCLUDED — an SFT row is an imitation target",
268
+ verifiers: "INCLUDED, reward forced to 0, flagged",
269
+ rft: "INCLUDED, reward forced to 0, flagged",
270
+ raw: "INCLUDED verbatim, flagged"
271
+ };
272
+ function gateSection(formats, gate, totalLines) {
273
+ const rows = formats.map((format) => {
274
+ const counts = gate.byFormat[format];
275
+ return [
276
+ format,
277
+ GATE_POLICY_LABEL[format],
278
+ String(counts?.emitted ?? 0),
279
+ String(counts?.excluded ?? 0),
280
+ counts?.maxEmittedReward === null || counts === void 0 ? "—" : formatReward(counts.maxEmittedReward),
281
+ counts?.maxEmittedEvidence == null ? "—" : `${counts.maxEmittedEvidence.path} = ${counts.maxEmittedEvidence.value}`,
282
+ GATE_FLAG_FIELD[format]
283
+ ];
284
+ });
285
+ const mixedExclusion = formats.some((format) => FORMAT_GATE_DISPOSITION[format] === "zero-and-flag");
286
+ return [
287
+ `\`outcome.realness_gated: true\` marks a run whose success signal was faked (it satisfied the proxy without doing the work). **${gate.gatedLines} of ${totalLines} lines in this release are gated.**`,
288
+ "",
289
+ "The gate is enforced, not asserted. A line pairing `realness_gated: true` with a reward above 0 is rejected by the schema validator, so it cannot be read out of a source ledger or written into `raw/train.jsonl` at all; on a gated line the numbers the reward was computed from (`outcome.metrics`, the verbatim judge `outcome.verdict`, and any per-step field the schema does not declare — a per-step reward is the same signal in credit-assignment form) are moved out of `outcome` and `steps[]` and into `provenance.gated_evidence`, where no config reads them as training input; the exporters re-check the same invariant on every line; and the release build measures the rows it is about to write and refuses to write ANY file if a gated row carries a positive reward — or any positive number derived from one — in ANY config.",
290
+ "",
291
+ "The table below is **measured on the rows in this release**, not a description of intent — the build counts the emitted rows and this card renders those counts:",
292
+ "",
293
+ markdownTable([
294
+ "config",
295
+ "policy",
296
+ "gated rows written",
297
+ "gated rows not written",
298
+ "max reward",
299
+ "max reward-derived number",
300
+ "flag"
301
+ ], rows),
302
+ "",
303
+ "Why gated rows are kept where they are kept: an SFT row is imitated verbatim, so a gamed trajectory must never appear in one at any weight. In `verifiers` the reward is a signed learning signal, and a gamed trajectory at reward 0 is a correct negative — dropping it would bias the negative population toward honest failures and leave a trainer no example of gaming being penalized. `rft` re-samples the completion, so only the prompt and the grader reference ship. `raw` is a faithful audit dump, where the gated row is the one an auditor most wants.",
304
+ "",
305
+ "Reward 0 is never the only label: every included config carries the flag on the row itself, because zeroing alone makes a faked success indistinguishable from an honest failure. Filter on the flag to drop the gamed population, or select on it to mine it. The gated run's own measurements are not destroyed either — they are parked verbatim under `provenance.gated_evidence` in `raw/train.jsonl`, which is where an auditor can see what the run claimed and why it was flagged.",
306
+ mixedExclusion ? "\n\"Gated rows not written\" is not all gate: `verifiers` also drops gap lines (empty transcript) and `rft` drops lines with no prompt turn, so that column can mix both causes." : ""
307
+ ].join("\n").trimEnd();
308
+ }
309
+ function roleRewardRows(lines) {
310
+ const counts = /* @__PURE__ */ new Map();
311
+ for (const line of lines) {
312
+ const byReward = counts.get(line.role) ?? /* @__PURE__ */ new Map();
313
+ const key = formatReward(line.outcome.reward);
314
+ byReward.set(key, (byReward.get(key) ?? 0) + 1);
315
+ counts.set(line.role, byReward);
316
+ }
317
+ const rows = [];
318
+ for (const role of ROLLOUT_ROLES) {
319
+ const byReward = counts.get(role);
320
+ if (!byReward) continue;
321
+ const keys = [...byReward.keys()].sort((a, b) => {
322
+ if (a === "null") return 1;
323
+ if (b === "null") return -1;
324
+ return Number(b) - Number(a);
325
+ });
326
+ for (const key of keys) rows.push([
327
+ role,
328
+ key,
329
+ String(byReward.get(key))
330
+ ]);
331
+ }
332
+ return rows;
333
+ }
334
+ function buildDatasetCard(inputs) {
335
+ const { lines, formats, includeProposers, sourceFiles, scrubTotals, excluded, formatCounts, gate } = inputs;
336
+ assertGateReport(gate);
337
+ const runIds = unique(lines.map((line) => line.run_id));
338
+ const generations = [...new Set(lines.map((line) => line.generation))].filter((g) => g !== null).sort((a, b) => a - b);
339
+ const models = unique(lines.map((line) => line.policy.model));
340
+ const harnesses = unique(lines.map((line) => line.policy.harness));
341
+ const captures = unique(lines.map((line) => line.provenance.capture));
342
+ const rewardSources = unique(lines.map((line) => line.outcome.reward_source));
343
+ const gapLines = lines.filter((line) => line.messages.length === 0).length;
344
+ const gatedLines = lines.filter((line) => line.outcome.realness_gated).length;
345
+ if (gatedLines !== gate.gatedLines) throw new Error(`dataset card: gate report claims ${gate.gatedLines} realness-gated line(s) but the lines it describes contain ${gatedLines} — the card and the build measured different data.`);
346
+ const frontmatter = [
347
+ "---",
348
+ "license: unknown",
349
+ "pretty_name: Tangle rollout ledger — agent trajectories",
350
+ "configs:",
351
+ formats.map((format) => [
352
+ ` - config_name: ${format}`,
353
+ " data_files:",
354
+ " - split: train",
355
+ ` path: ${FORMAT_FILES[format]}`
356
+ ].join("\n")).join("\n"),
357
+ "---"
358
+ ].join("\n");
359
+ const formatsTable = markdownTable([
360
+ "config",
361
+ "path",
362
+ "rows",
363
+ "contents"
364
+ ], formats.map((format) => [
365
+ format,
366
+ `\`${FORMAT_FILES[format]}\``,
367
+ String(formatCounts[format] ?? 0),
368
+ FORMAT_DESCRIPTIONS[format]
369
+ ]));
370
+ const countsTable = markdownTable([
371
+ "role",
372
+ "reward",
373
+ "lines"
374
+ ], roleRewardRows(lines));
375
+ const scrubTable = markdownTable(["rule", "rewrites"], Object.entries(scrubTotals).map(([rule, count]) => [rule, String(count)]));
376
+ const proposerNote = includeProposers ? "Proposer sessions are INCLUDED (`--include-proposers`); their transcripts contain improvement-loop harness source." : `Proposer sessions are excluded by default (${excluded.proposers} lines dropped); they contain improvement-loop harness source. Rebuild with \`--include-proposers\` to keep them.`;
377
+ return `${frontmatter}
378
+
379
+ # Tangle rollout ledger — agent trajectories
380
+
381
+ One line per agent invocation (supervisor episode, worker session, proposer shot, judge call, analyst pass) captured by the \`${ROLLOUT_SCHEMA}\` rollout ledger, labeled with improvement-loop coordinates and the official-judge reward, with the full message transcript inline.
382
+
383
+ This release contains the **trainable split only** (\`search\`). Holdout, dev, and canary splits are structurally excluded at build time, and the build additionally drops any non-trainable line as a fail-closed filter (${excluded.nonTrain} dropped here).
384
+
385
+ ## Formats
386
+
387
+ ${formatsTable}
388
+
389
+ ## Anti-Goodhart gate (\`realness_gated\`)
390
+
391
+ ${gateSection(formats, gate, lines.length)}
392
+
393
+ ## Schema (\`${ROLLOUT_SCHEMA}\`)
394
+
395
+ Each raw line carries:
396
+
397
+ - \`rollout_id\` / \`parent_rollout_id\` — invocation identity; workers point at their spawning supervisor episode.
398
+ - \`run_id\`, \`experiment_id\`, \`candidate_id\` — run/experiment/candidate identity from the producing RunRecord, when present.
399
+ - \`generation\`, \`candidate_index\` — improvement-loop coordinates (\`-1\` = baseline campaign; \`null\` = not an improvement loop).
400
+ - \`role\` — one of ${ROLLOUT_ROLES.map((role) => `\`${role}\``).join(", ")}.
401
+ - \`task\` — suite, instance id, split, seed, replicate index.
402
+ - \`policy\` — harness, model, provider, profile commit, prompt/config hashes, sampling params.
403
+ - \`messages\` / \`tool_defs\` — full transcript in canonical OpenAI chat-with-tools form (including \`reasoning_content\`). An empty \`messages\` array is a labeled gap line; \`provenance.gap\` says why the transcript could not be recovered (${gapLines} gap lines in this release).
404
+ - \`outcome\` — \`reward\` (the single scalar), \`reward_source\`, the verbatim judge \`verdict\`, non-scalar \`metrics\`, and \`realness_gated\` (anti-Goodhart flag; see [Anti-Goodhart gate](#anti-goodhart-gate-realness_gated) for the per-config treatment and the measured counts).
405
+ - \`cost\` — usd, token counts, wall time.
406
+ - \`artifacts\` / \`provenance\` — patch/run-dir/transcript pointers (scrubbed) and capture metadata.
407
+
408
+ ## Provenance
409
+
410
+ - Source ledgers: ${sourceFiles.map((file) => `\`${file}\``).join(", ")}
411
+ - Run ids: ${runIds.map((id) => `\`${id}\``).join(", ")}
412
+ - Generations: ${generations.join(", ")} (\`-1\` = baseline campaign)
413
+ - Models: ${models.map((m) => `\`${m}\``).join(", ")}
414
+ - Harnesses: ${harnesses.map((h) => `\`${h}\``).join(", ")}
415
+ - Capture modes: ${captures.join(", ")}
416
+ - Every reward traces to a named source (reward sources in this release: ${rewardSources.map((s) => `\`${s}\``).join(", ")}).
417
+
418
+ ${proposerNote}
419
+
420
+ ## Reward semantics per role
421
+
422
+ - **agent** — the producing RunRecord's holdout/search score, with the realness gate forcing gamed successes to 0.
423
+ - **supervisor** — the official-judge verdict on the episode's delivered artifact (1 = resolved, 0 = not).
424
+ - **worker** — INHERITED from the parent supervisor episode (\`…/inherited\`). Caveat: reward 1 does not establish this worker's individual contribution (sibling workers in the same episode share the episode outcome), and reward 0 does not prove this worker failed.
425
+ - **proposer** — the fraction of improvement-set instances the proposed candidate resolved (\`…/candidate-resolved-fraction\`); a scalar in [0, 1], not a binary verdict.
426
+ - **judge / analyst** — carry the episode verdict where one applies; otherwise \`reward: null\` (a labeled gap, never 0).
427
+
428
+ ## Counts
429
+
430
+ ${countsTable}
431
+
432
+ Total lines: ${lines.length}
433
+
434
+ ## Scrubbing
435
+
436
+ Absolute home paths were rewritten to \`$WORK\`, credential-shaped strings were replaced with \`[REDACTED:<kind>]\` markers, internal hostnames were normalized to \`*.internal.example\`, and username-bearing incidentals (\`ls -l\` owner columns, per-user pytest tmpdirs) were normalized to \`user\` / \`$USER\`. Rewrite counts for this release (full per-file breakdown in \`scrub-report.json\`):
437
+
438
+ ${scrubTable}
439
+
440
+ ## License
441
+
442
+ \`license: unknown\` is a placeholder — the releasing operator must set the real SPDX license id in the frontmatter above before publishing.
443
+
444
+ ## Citation
445
+
446
+ \`\`\`bibtex
447
+ @misc{tangle_rollout_ledger,
448
+ title = {Tangle rollout ledger — agent trajectories},
449
+ author = {{Tangle Network}},
450
+ howpublished = {HuggingFace Datasets},
451
+ note = {Operator: fill in the repository URL, authors, and year before publishing}
452
+ }
453
+ \`\`\`
454
+ `;
455
+ }
456
+ //#endregion
457
+ //#region src/rollout/release/scrub.ts
458
+ /**
459
+ * Deterministic scrubbing pass over rollout-ledger lines before public release.
460
+ *
461
+ * Every rule is a pure regex rewrite applied to every string value in a line
462
+ * (messages, artifacts, run ids, tool arguments — everywhere), so the scrubbed
463
+ * line is still a valid `tangle.rollout.v1` line. Rules are idempotent:
464
+ * scrub(scrub(x)) === scrub(x), and a second pass counts zero hits — that is
465
+ * the property the release pipeline relies on to prove nothing half-scrubbed
466
+ * ships. Rule order matters: whole `KEY=value` env pairs are redacted before
467
+ * the bare-key rule so one secret is never counted twice.
468
+ */
469
+ const SCRUB_RULES = [
470
+ {
471
+ name: "home-path",
472
+ pattern: /\/(?:home|Users)\/[A-Za-z0-9._-]+/g,
473
+ rewrite: () => "$WORK"
474
+ },
475
+ {
476
+ name: "home-path-encoded",
477
+ pattern: /(?<=\/)-(?:home|Users)-[A-Za-z0-9_.]+/g,
478
+ rewrite: () => "$WORK"
479
+ },
480
+ {
481
+ name: "tmp-user-dir",
482
+ pattern: /\/tmp\/pytest-of-[A-Za-z0-9._-]+/g,
483
+ rewrite: () => "/tmp/pytest-of-$USER"
484
+ },
485
+ {
486
+ name: "ls-owner",
487
+ pattern: /([-bcdlps][-rwxsStT]{9}[.+@]?\s+\d+\s+)(?!user user(?=\s))[A-Za-z0-9._-]+\s+[A-Za-z0-9._-]+(?=\s)/g,
488
+ rewrite: (_match, prefix) => `${prefix}user user`
489
+ },
490
+ {
491
+ name: "env-secret",
492
+ pattern: /\b([A-Z][A-Z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIALS?))=(?!\[REDACTED:)("[^"]*"|'[^']*'|[^\s"']+)/g,
493
+ rewrite: (_match, name) => `${name}=[REDACTED:env]`
494
+ },
495
+ {
496
+ name: "bearer-token",
497
+ pattern: /\b(Bearer|Basic)\s+(?!\[REDACTED:)[A-Za-z0-9\-._~+/=]{8,}/g,
498
+ rewrite: (_match, scheme) => `${scheme} [REDACTED:bearer]`
499
+ },
500
+ {
501
+ name: "api-key",
502
+ pattern: /\b(?:sk-[A-Za-z0-9_-]{16,}|gh[pousr]_[A-Za-z0-9]{16,}|github_pat_[A-Za-z0-9_]{20,}|hf_[A-Za-z0-9]{16,}|xox[baprs]-[A-Za-z0-9-]{10,}|AKIA[0-9A-Z]{16})\b/g,
503
+ rewrite: () => "[REDACTED:api-key]"
504
+ },
505
+ {
506
+ name: "infra-host",
507
+ pattern: /(?<![A-Za-z0-9.-])((?:[A-Za-z0-9-]+\.)*)tangle\.(?:tools|network)(?![A-Za-z0-9-])/g,
508
+ rewrite: (_match, prefix) => `${prefix ?? ""}internal.example`
509
+ },
510
+ {
511
+ name: "machine-host",
512
+ pattern: /\b[A-Za-z0-9]+-GTR-Pro\b/g,
513
+ rewrite: () => "workstation"
514
+ }
515
+ ];
516
+ function emptyScrubCounts() {
517
+ return Object.fromEntries(SCRUB_RULES.map((rule) => [rule.name, 0]));
518
+ }
519
+ function addScrubCounts(into, from) {
520
+ for (const [name, count] of Object.entries(from)) into[name] = (into[name] ?? 0) + count;
521
+ return into;
522
+ }
523
+ function scrubText(text, counts) {
524
+ let out = text;
525
+ for (const rule of SCRUB_RULES) out = out.replace(rule.pattern, (match, ...rest) => {
526
+ counts[rule.name] = (counts[rule.name] ?? 0) + 1;
527
+ return rule.rewrite(match, typeof rest[0] === "string" ? rest[0] : void 0);
528
+ });
529
+ return out;
530
+ }
531
+ function scrubValue(value, counts) {
532
+ if (typeof value === "string") return scrubText(value, counts);
533
+ if (Array.isArray(value)) return value.map((item) => scrubValue(item, counts));
534
+ if (value !== null && typeof value === "object") {
535
+ const out = {};
536
+ for (const [key, item] of Object.entries(value)) out[key] = scrubValue(item, counts);
537
+ return out;
538
+ }
539
+ return value;
540
+ }
541
+ /**
542
+ * Scrub every string value in a line; structure and key order are preserved.
543
+ *
544
+ * `assertMinted` on the way out rather than a cast: scrubbing rebuilds the
545
+ * object, so the brand has to be re-earned, and re-validating proves the rules
546
+ * did not rewrite a field the schema constrains (`reward` is a number, not a
547
+ * string, so no rule should ever touch it — this is what checks that).
548
+ */
549
+ function scrubRolloutLine(line, counts) {
550
+ return assertMinted(scrubValue(line, counts), `scrubbed rollout line ${line.rollout_id}`);
551
+ }
552
+ function scrubLines(lines) {
553
+ const counts = emptyScrubCounts();
554
+ return {
555
+ lines: lines.map((line) => scrubRolloutLine(line, counts)),
556
+ counts
557
+ };
558
+ }
559
+ /**
560
+ * A `RolloutScrubber` (text → text) applying the full rule set — the
561
+ * default hook to pass to `mintRolloutRows({ scrub })` so lines are
562
+ * scrubbed at mint time, before they ever reach a ledger file. Release
563
+ * builds re-run `scrubLines` regardless (idempotent), so double-scrubbing
564
+ * is safe and counted as zero.
565
+ */
566
+ function defaultRolloutScrubber(text) {
567
+ return scrubText(text, emptyScrubCounts());
568
+ }
569
+ //#endregion
570
+ //#region src/rollout/release/hf-dataset.ts
571
+ /**
572
+ * One-command HuggingFace dataset release from rollout ledgers:
573
+ *
574
+ * agent-eval rollout-release <ledger.jsonl...> --out <dir> \
575
+ * [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]
576
+ *
577
+ * Pipeline per input ledger: read + validate → fail-closed filters
578
+ * (trainable split only; proposer sessions dropped unless
579
+ * --include-proposers, they contain improvement-loop harness source) →
580
+ * deterministic scrub → export the requested formats + scrub-report.json +
581
+ * auto-generated README.md card. Deterministic: same inputs and flags →
582
+ * byte-identical output dir.
583
+ *
584
+ * --push uploads the built dir with `huggingface-cli upload` only when the
585
+ * CLI exists on PATH and HF_TOKEN is present in the env; the token is
586
+ * never printed. Everything else runs fully offline.
587
+ */
588
+ async function buildHfDataset(inputs, options) {
589
+ if (inputs.length === 0) throw new Error("no input ledgers given");
590
+ if (options.formats.length === 0) throw new Error("no formats selected");
591
+ const report = {
592
+ files: {},
593
+ totals: emptyScrubCounts(),
594
+ excluded: {
595
+ proposers: 0,
596
+ nonTrain: 0
597
+ }
598
+ };
599
+ const kept = [];
600
+ let read = 0;
601
+ for (const input of inputs) {
602
+ const lines = await readRolloutLedger(input);
603
+ read += lines.length;
604
+ const scrubbed = scrubLines(lines.filter((line) => {
605
+ if (!isTrainableSplit(line.task.split)) {
606
+ report.excluded.nonTrain += 1;
607
+ return false;
608
+ }
609
+ if (!options.includeProposers && line.role === "proposer") {
610
+ report.excluded.proposers += 1;
611
+ return false;
612
+ }
613
+ return true;
614
+ }));
615
+ report.files[input] = scrubbed.counts;
616
+ addScrubCounts(report.totals, scrubbed.counts);
617
+ kept.push(...scrubbed.lines);
618
+ }
619
+ const formatCounts = {};
620
+ const files = [];
621
+ const gated = gatedRolloutIds(kept);
622
+ const gate = {
623
+ gatedLines: gated.size,
624
+ byFormat: {}
625
+ };
626
+ const pending = [];
627
+ for (const format of options.formats) {
628
+ const path = join(options.out, FORMAT_FILES[format]);
629
+ if (format === "raw") {
630
+ gate.byFormat.raw = measureFormatGate(gated, releaseRowRefs.raw(kept));
631
+ formatCounts.raw = kept.length;
632
+ pending.push({
633
+ path,
634
+ write: () => writeRolloutLedger(path, kept)
635
+ });
636
+ } else if (format === "sft") {
637
+ const rows = toSftRows(kept);
638
+ gate.byFormat.sft = measureFormatGate(gated, releaseRowRefs.sft(rows));
639
+ formatCounts.sft = rows.length;
640
+ pending.push({
641
+ path,
642
+ write: () => writeFile(path, toJsonl(rows))
643
+ });
644
+ } else if (format === "verifiers") {
645
+ const outputs = toVerifiersRolloutOutputs(kept, { gatedLines: FORMAT_GATE_DISPOSITION.verifiers });
646
+ gate.byFormat.verifiers = measureFormatGate(gated, releaseRowRefs.verifiers(outputs));
647
+ formatCounts.verifiers = outputs.length;
648
+ pending.push({
649
+ path,
650
+ write: () => writeFile(path, toJsonl(outputs))
651
+ });
652
+ } else {
653
+ const items = toRftItems(kept, { gatedLines: FORMAT_GATE_DISPOSITION.rft });
654
+ gate.byFormat.rft = measureFormatGate(gated, releaseRowRefs.rft(items));
655
+ formatCounts.rft = items.length;
656
+ pending.push({
657
+ path,
658
+ write: () => writeFile(path, toJsonl(items))
659
+ });
660
+ }
661
+ }
662
+ assertGateReport(gate);
663
+ for (const { path, write } of pending) {
664
+ await mkdir(dirname(path), { recursive: true });
665
+ await write();
666
+ files.push(path);
667
+ }
668
+ const reportPath = join(options.out, "scrub-report.json");
669
+ await writeFile(reportPath, `${JSON.stringify(report, null, 2)}\n`);
670
+ files.push(reportPath);
671
+ const cardPath = join(options.out, "README.md");
672
+ await writeFile(cardPath, buildDatasetCard({
673
+ lines: kept,
674
+ formats: options.formats,
675
+ includeProposers: options.includeProposers,
676
+ sourceFiles: inputs.map((input) => basename(input)),
677
+ scrubTotals: report.totals,
678
+ excluded: report.excluded,
679
+ formatCounts,
680
+ gate
681
+ }));
682
+ files.push(cardPath);
683
+ return {
684
+ inputs,
685
+ read,
686
+ kept: kept.length,
687
+ scrub: report,
688
+ formatCounts,
689
+ gate,
690
+ files
691
+ };
692
+ }
693
+ function planPushCommand(repo, outDir) {
694
+ return [
695
+ "huggingface-cli",
696
+ "upload",
697
+ repo,
698
+ outDir,
699
+ ".",
700
+ "--repo-type",
701
+ "dataset"
702
+ ];
703
+ }
704
+ function pushDataset(repo, outDir) {
705
+ if (spawnSync("which", ["huggingface-cli"], { stdio: "ignore" }).status !== 0) throw new Error("huggingface-cli not found on PATH — install huggingface_hub[cli] before --push");
706
+ if (!process.env.HF_TOKEN) throw new Error("HF_TOKEN not present in env — refusing to push");
707
+ const [command, ...args] = planPushCommand(repo, outDir);
708
+ const run = spawnSync(command, args, { stdio: "inherit" });
709
+ if (run.status !== 0) throw new Error(`huggingface-cli upload exited ${String(run.status)}`);
710
+ }
711
+ const ROLLOUT_RELEASE_USAGE = "usage: agent-eval rollout-release <ledger.jsonl...> --out <dir> [--formats sft,verifiers,rft,raw] [--include-proposers] [--push <org/name>]";
712
+ function parseRolloutReleaseArgs(argv) {
713
+ const args = {
714
+ inputs: [],
715
+ out: "",
716
+ formats: [...RELEASE_FORMATS],
717
+ includeProposers: false,
718
+ push: null
719
+ };
720
+ for (let i = 0; i < argv.length; i++) {
721
+ const arg = argv[i];
722
+ if (arg === "--out") args.out = argv[++i] ?? "";
723
+ else if (arg === "--formats") {
724
+ const raw = (argv[++i] ?? "").split(",").filter(Boolean);
725
+ for (const format of raw) if (!RELEASE_FORMATS.includes(format)) throw new Error(`unknown format "${format}" — expected one of ${RELEASE_FORMATS.join(",")}`);
726
+ args.formats = raw;
727
+ } else if (arg === "--include-proposers") args.includeProposers = true;
728
+ else if (arg === "--push") args.push = argv[++i] ?? null;
729
+ else if (arg.startsWith("--")) throw new Error(`unknown flag "${arg}"`);
730
+ else args.inputs.push(arg);
731
+ }
732
+ if (args.inputs.length === 0 || !args.out) throw new Error(ROLLOUT_RELEASE_USAGE);
733
+ if (args.push !== null && !/^[\w.-]+\/[\w.-]+$/.test(args.push)) throw new Error(`--push expects <org/name>, got "${args.push}"`);
734
+ return args;
735
+ }
736
+ /** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
737
+ async function runRolloutReleaseCli(argv) {
738
+ let args;
739
+ try {
740
+ args = parseRolloutReleaseArgs(argv);
741
+ } catch (error) {
742
+ process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
743
+ return 2;
744
+ }
745
+ const summary = await buildHfDataset(args.inputs, args);
746
+ process.stdout.write(`${JSON.stringify({
747
+ read: summary.read,
748
+ kept: summary.kept,
749
+ formatCounts: summary.formatCounts,
750
+ gate: summary.gate,
751
+ scrub: summary.scrub
752
+ }, null, 2)}\n`);
753
+ process.stdout.write(`dataset → ${args.out} (${summary.files.length} files)\n`);
754
+ if (args.push !== null) {
755
+ pushDataset(args.push, args.out);
756
+ process.stdout.write(`pushed → ${args.push}\n`);
757
+ }
758
+ return 0;
759
+ }
760
+ //#endregion
761
+ export { readRolloutJournal as C, appendRolloutLines as S, writeRolloutLedger as T, FORMAT_GATE_DISPOSITION as _, pushDataset as a, measureFormatGate as b, addScrubCounts as c, scrubLines as d, scrubRolloutLine as f, buildDatasetCard as g, RELEASE_FORMATS as h, planPushCommand as i, defaultRolloutScrubber as l, FORMAT_FILES as m, buildHfDataset as n, runRolloutReleaseCli as o, scrubText as p, parseRolloutReleaseArgs as r, SCRUB_RULES as s, ROLLOUT_RELEASE_USAGE as t, emptyScrubCounts as u, assertGateReport as v, readRolloutLedger as w, releaseRowRefs as x, gatedRolloutIds as y };
762
+
763
+ //# sourceMappingURL=hf-dataset-DBJXXoY1.js.map