@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
package/dist/rl.js CHANGED
@@ -1,2077 +1,2444 @@
1
- import {
2
- doublyRobust,
3
- inverseProbabilityWeighting,
4
- offPolicyEstimateAll,
5
- selfNormalizedImportanceWeighting
6
- } from "./chunk-T6RLYGAD.js";
7
- import {
8
- FileSystemOutcomeStore,
9
- InMemoryOutcomeStore
10
- } from "./chunk-3RF76KTD.js";
11
- import {
12
- runEvalCampaign
13
- } from "./chunk-YQN4ICPP.js";
14
- import {
15
- mintRolloutRows
16
- } from "./chunk-H23X7XKK.js";
17
- import {
18
- isSplitEligible
19
- } from "./chunk-OWN5NPMC.js";
20
- import {
21
- assertRewardGate
22
- } from "./chunk-PC5DOSM7.js";
23
- import "./chunk-RZTMDUO7.js";
24
- import {
25
- detectRewardHacking,
26
- extractVerifiableReward,
27
- extractVerifiableRewardsFromRecords,
28
- filterDeterministicallyRewarded
29
- } from "./chunk-EG66UGL4.js";
30
- import {
31
- campaignCellToRunRecord
32
- } from "./chunk-E7QXT7SX.js";
33
- import "./chunk-SFLLL76A.js";
34
- import {
35
- rubricPredictiveValidity
36
- } from "./chunk-TQ7LNKZ3.js";
37
- import {
38
- evaluateInterimReleaseConfidence
39
- } from "./chunk-MAZ26DC7.js";
40
- import "./chunk-TJVT4QFF.js";
41
- import "./chunk-7FO3TNPI.js";
42
- import {
43
- benjaminiHochberg,
44
- wilcoxonSignedRank
45
- } from "./chunk-ZHTZ4EYI.js";
46
- import {
47
- observationsFromRunRecords,
48
- thompsonCurriculum,
49
- varianceBasedCurriculum
50
- } from "./chunk-G7MGMCZD.js";
51
- import "./chunk-VCZ5FQYW.js";
52
- import "./chunk-VI2UW6B6.js";
53
- import {
54
- InMemoryTraceStore
55
- } from "./chunk-U4PHLT2N.js";
56
- import "./chunk-PC4UYEBM.js";
57
- import "./chunk-VQMK5FMP.js";
58
- import {
59
- runTaskScore
60
- } from "./chunk-56TAVBOK.js";
61
- import "./chunk-MA6HLL3S.js";
62
- import {
63
- observedSplitScore,
64
- trainingScore
65
- } from "./chunk-OIUOT4QD.js";
66
- import {
67
- ValidationError
68
- } from "./chunk-ONWEPEDO.js";
69
- import "./chunk-PZ5AY32C.js";
70
-
71
- // src/rl/adaptation-eval.ts
1
+ import { s as ValidationError } from "./errors-8YnH8WlF.js";
2
+ import { o as runTaskScore } from "./run-record-BuoE80Dq.js";
3
+ import { N as wilcoxonSignedRank, t as benjaminiHochberg } from "./statistics-CnnxdpOg.js";
4
+ import { r as observedSplitScore, s as trainingScore } from "./reward-nw2xZGZG.js";
5
+ import { l as assertRewardGate } from "./schema-C6DW4ZHR.js";
6
+ import { t as isSplitEligible } from "./exporters-q9iL-2Jf.js";
7
+ import { t as mintRolloutRows } from "./mint-yN2M2eh0.js";
8
+ import { a as InMemoryTraceStore } from "./integrity-BzRbCHzi.js";
9
+ import { c as campaignCellToRunRecord, i as filterDeterministicallyRewarded, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-qipEpKvY.js";
10
+ import { t as runEvalCampaign } from "./eval-campaign-DEm6c8ru.js";
11
+ import { t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
12
+ import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-B3xmbmS1.js";
13
+ import { n as thompsonCurriculum, r as varianceBasedCurriculum, t as observationsFromRunRecords } from "./active-curriculum-C4mk67HP.js";
14
+ import { i as selfNormalizedImportanceWeighting, n as inverseProbabilityWeighting, r as offPolicyEstimateAll, t as doublyRobust } from "./off-policy-DvgzvtIx.js";
15
+ import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-ChBKlTd_.js";
16
+ import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
17
+ import { dirname } from "node:path";
18
+ //#region src/rl/adaptation-eval.ts
72
19
  async function runAdaptationCurve(opts) {
73
- const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
74
- const reps = opts.reps ?? 3;
75
- const passThreshold = opts.passThreshold ?? 0.5;
76
- const sortedKs = [...ks].sort((a, b) => a - b);
77
- const points = [];
78
- for (const k of sortedKs) {
79
- const perScenario = [];
80
- const allScores = [];
81
- let totalPasses = 0;
82
- let totalAttempts = 0;
83
- for (const scenario of opts.scenarios) {
84
- const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
85
- const scores = [];
86
- let passes = 0;
87
- for (let r = 0; r < reps; r++) {
88
- const score = await opts.runner.run({ scenario, k, rep: r });
89
- scores.push(score);
90
- if (score >= passThreshold) passes++;
91
- allScores.push(score);
92
- if (score >= passThreshold) totalPasses++;
93
- totalAttempts++;
94
- }
95
- const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
96
- perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
97
- }
98
- const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
99
- const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
100
- points.push({
101
- k,
102
- meanScore,
103
- passRate: totalPasses / Math.max(1, totalAttempts),
104
- std: Math.sqrt(variance),
105
- n: allScores.length,
106
- perScenario
107
- });
108
- }
109
- const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
110
- const maxK = sortedKs[sortedKs.length - 1] ?? 1;
111
- let area = 0;
112
- for (let i = 1; i < points.length; i++) {
113
- const x1 = points[i - 1].k;
114
- const x2 = points[i].k;
115
- const y1 = points[i - 1].meanScore;
116
- const y2 = points[i].meanScore;
117
- area += (y1 + y2) / 2 * (x2 - x1);
118
- }
119
- const adaptationArea = maxK === 0 ? 0 : area / maxK;
120
- return { points, firstPassK: firstPassK2, adaptationArea };
121
- }
20
+ const ks = opts.ks ?? [
21
+ 0,
22
+ 1,
23
+ 2,
24
+ 4,
25
+ 8,
26
+ 16
27
+ ];
28
+ const reps = opts.reps ?? 3;
29
+ const passThreshold = opts.passThreshold ?? .5;
30
+ const sortedKs = [...ks].sort((a, b) => a - b);
31
+ const points = [];
32
+ for (const k of sortedKs) {
33
+ const perScenario = [];
34
+ const allScores = [];
35
+ let totalPasses = 0;
36
+ let totalAttempts = 0;
37
+ for (const scenario of opts.scenarios) {
38
+ const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
39
+ const scores = [];
40
+ let passes = 0;
41
+ for (let r = 0; r < reps; r++) {
42
+ const score = await opts.runner.run({
43
+ scenario,
44
+ k,
45
+ rep: r
46
+ });
47
+ scores.push(score);
48
+ if (score >= passThreshold) passes++;
49
+ allScores.push(score);
50
+ if (score >= passThreshold) totalPasses++;
51
+ totalAttempts++;
52
+ }
53
+ const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
54
+ perScenario.push({
55
+ scenarioId: sid,
56
+ meanScore: meanS,
57
+ passes,
58
+ total: scores.length
59
+ });
60
+ }
61
+ const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
62
+ const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
63
+ points.push({
64
+ k,
65
+ meanScore,
66
+ passRate: totalPasses / Math.max(1, totalAttempts),
67
+ std: Math.sqrt(variance),
68
+ n: allScores.length,
69
+ perScenario
70
+ });
71
+ }
72
+ const firstPassK = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
73
+ const maxK = sortedKs[sortedKs.length - 1] ?? 1;
74
+ let area = 0;
75
+ for (let i = 1; i < points.length; i++) {
76
+ const x1 = points[i - 1].k;
77
+ const x2 = points[i].k;
78
+ const y1 = points[i - 1].meanScore;
79
+ const y2 = points[i].meanScore;
80
+ area += (y1 + y2) / 2 * (x2 - x1);
81
+ }
82
+ return {
83
+ points,
84
+ firstPassK,
85
+ adaptationArea: maxK === 0 ? 0 : area / maxK
86
+ };
87
+ }
88
+ /**
89
+ * Paired comparison of two adaptation curves. Per-k deltas with 95%
90
+ * bootstrap CIs (constructed from each curve's `perScenario` per-k means
91
+ * — the bootstrap unit is the scenario, not the rep).
92
+ */
122
93
  function compareAdaptationCurves(a, b, opts = {}) {
123
- const conf = opts.confidence ?? 0.95;
124
- const resamples = opts.bootstrapResamples ?? 500;
125
- const rng = makeRng(opts.seed);
126
- const perK = [];
127
- for (const ap of a.points) {
128
- const bp = b.points.find((p) => p.k === ap.k);
129
- if (!bp) continue;
130
- const aMeans = ap.perScenario.map((s) => s.meanScore);
131
- const bMeans = bp.perScenario.map((s) => s.meanScore);
132
- const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
133
- const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
134
- perK.push({
135
- k: ap.k,
136
- deltaMean: ap.meanScore - bp.meanScore,
137
- aLow: aCi.low,
138
- aHigh: aCi.high,
139
- bLow: bCi.low,
140
- bHigh: bCi.high
141
- });
142
- }
143
- const areaDelta = a.adaptationArea - b.adaptationArea;
144
- const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
145
- const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
146
- let verdict;
147
- if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
148
- else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
149
- else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
150
- else verdict = "similar";
151
- const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
152
- return { perK, areaDelta, firstPassKDelta, verdict, rationale };
153
- }
154
- function firstPassK(curve, threshold = 0.5) {
155
- return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
94
+ const conf = opts.confidence ?? .95;
95
+ const resamples = opts.bootstrapResamples ?? 500;
96
+ const rng = makeRng(opts.seed);
97
+ const perK = [];
98
+ for (const ap of a.points) {
99
+ const bp = b.points.find((p) => p.k === ap.k);
100
+ if (!bp) continue;
101
+ const aMeans = ap.perScenario.map((s) => s.meanScore);
102
+ const bMeans = bp.perScenario.map((s) => s.meanScore);
103
+ const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
104
+ const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
105
+ perK.push({
106
+ k: ap.k,
107
+ deltaMean: ap.meanScore - bp.meanScore,
108
+ aLow: aCi.low,
109
+ aHigh: aCi.high,
110
+ bLow: bCi.low,
111
+ bHigh: bCi.high
112
+ });
113
+ }
114
+ const areaDelta = a.adaptationArea - b.adaptationArea;
115
+ const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
116
+ const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
117
+ let verdict;
118
+ if (Math.abs(meanDelta) < .02 && Math.abs(areaDelta) < .02) verdict = "similar";
119
+ else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
120
+ else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
121
+ else verdict = "similar";
122
+ const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
123
+ return {
124
+ perK,
125
+ areaDelta,
126
+ firstPassKDelta,
127
+ verdict,
128
+ rationale
129
+ };
130
+ }
131
+ /** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
132
+ function firstPassK(curve, threshold = .5) {
133
+ return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
156
134
  }
157
135
  function makeRng(seed) {
158
- if (seed === void 0) return Math.random;
159
- let s = seed >>> 0;
160
- return () => {
161
- s = s + 1831565813 >>> 0;
162
- let t = s;
163
- t = Math.imul(t ^ t >>> 15, t | 1);
164
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
165
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
166
- };
136
+ if (seed === void 0) return Math.random;
137
+ let s = seed >>> 0;
138
+ return () => {
139
+ s = s + 1831565813 >>> 0;
140
+ let t = s;
141
+ t = Math.imul(t ^ t >>> 15, t | 1);
142
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
143
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
144
+ };
167
145
  }
168
146
  function bootstrapMeanCi(xs, resamples, confidence, rng) {
169
- if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
170
- const samples = new Array(resamples);
171
- for (let b = 0; b < resamples; b++) {
172
- let sum = 0;
173
- for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
174
- samples[b] = sum / xs.length;
175
- }
176
- samples.sort((a, b) => a - b);
177
- const alpha = 1 - confidence;
178
- return {
179
- low: samples[Math.floor(alpha / 2 * resamples)],
180
- high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
181
- };
182
- }
183
-
184
- // src/rl/compute-curves.ts
147
+ if (xs.length < 2) return {
148
+ low: xs[0] ?? 0,
149
+ high: xs[0] ?? 0
150
+ };
151
+ const samples = new Array(resamples);
152
+ for (let b = 0; b < resamples; b++) {
153
+ let sum = 0;
154
+ for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
155
+ samples[b] = sum / xs.length;
156
+ }
157
+ samples.sort((a, b) => a - b);
158
+ const alpha = 1 - confidence;
159
+ return {
160
+ low: samples[Math.floor(alpha / 2 * resamples)],
161
+ high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
162
+ };
163
+ }
164
+ //#endregion
165
+ //#region src/rl/compute-curves.ts
166
+ /**
167
+ * Test-time compute scaling curves.
168
+ *
169
+ * The test-time-compute frontier paper (Snell et al. 2024) and the
170
+ * subsequent o1-style scaling work both show that LLM-agent capability
171
+ * is a function of the compute budget at inference, not just of the
172
+ * training run. The right way to characterize a candidate is therefore
173
+ * a *curve* — score at compute budgets {1×, 4×, 16×, …} — not a single
174
+ * point.
175
+ *
176
+ * This module ships:
177
+ *
178
+ * 1. The compute-curve harness — `runComputeCurve(runner, budgets)` —
179
+ * that evaluates one candidate at a sequence of compute budgets
180
+ * and returns the (compute, score) curve.
181
+ * 2. A best-of-N evaluator — `bestOfN(runner, n, scoreFn)` — the
182
+ * simplest test-time-compute scaling primitive: sample N
183
+ * independent rollouts, return the best.
184
+ * 3. A self-consistency evaluator — `selfConsistency(runner, n)` —
185
+ * the majority-vote variant of best-of-N for tasks with a small
186
+ * categorical answer space.
187
+ * 4. Pareto-frontier extraction over multiple candidates — given
188
+ * (candidate, compute, score) tuples, return the set of
189
+ * candidate-compute combinations that aren't dominated.
190
+ *
191
+ * Caveat: "compute" here is the caller's notion of a compute unit. For
192
+ * agent eval that's typically wall-time × parallelism, or token budget,
193
+ * or LLM-call count. We accept whatever the caller provides; the curve
194
+ * is on whatever axis they pick.
195
+ */
185
196
  async function runComputeCurve(opts) {
186
- const points = [];
187
- for (const budget of opts.budgets) {
188
- const r = await opts.runAtBudget(budget);
189
- points.push({
190
- budgetId: budget.id,
191
- cost: budget.cost,
192
- score: r.score,
193
- samples: r.samples,
194
- std: r.std,
195
- metrics: r.metrics
196
- });
197
- }
198
- const sorted = [...points].sort((a, b) => a.cost - b.cost);
199
- const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null;
200
- const best = points.reduce((a, b) => b.score > a.score ? b : a);
201
- return { candidateId: opts.candidateId, points: sorted, logSlope, best };
202
- }
197
+ const points = [];
198
+ for (const budget of opts.budgets) {
199
+ const r = await opts.runAtBudget(budget);
200
+ points.push({
201
+ budgetId: budget.id,
202
+ cost: budget.cost,
203
+ score: r.score,
204
+ samples: r.samples,
205
+ std: r.std,
206
+ metrics: r.metrics
207
+ });
208
+ }
209
+ const sorted = [...points].sort((a, b) => a.cost - b.cost);
210
+ const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null;
211
+ const best = points.reduce((a, b) => b.score > a.score ? b : a);
212
+ return {
213
+ candidateId: opts.candidateId,
214
+ points: sorted,
215
+ logSlope,
216
+ best
217
+ };
218
+ }
219
+ /** The simplest test-time scaling primitive. */
203
220
  async function bestOfN(opts) {
204
- if (opts.n <= 0) throw new ValidationError("bestOfN: n must be > 0");
205
- const rollouts = [];
206
- const scores = [];
207
- for (let i = 0; i < opts.n; i++) {
208
- const r = await opts.sample(i);
209
- rollouts.push(r);
210
- scores.push(await opts.scoreFn(r));
211
- }
212
- let bestIndex = 0;
213
- for (let i = 1; i < scores.length; i++) if (scores[i] > scores[bestIndex]) bestIndex = i;
214
- const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length;
215
- return {
216
- best: rollouts[bestIndex],
217
- bestScore: scores[bestIndex],
218
- scores,
219
- meanScore,
220
- bestIndex
221
- };
222
- }
221
+ if (opts.n <= 0) throw new ValidationError("bestOfN: n must be > 0");
222
+ const rollouts = [];
223
+ const scores = [];
224
+ for (let i = 0; i < opts.n; i++) {
225
+ const r = await opts.sample(i);
226
+ rollouts.push(r);
227
+ scores.push(await opts.scoreFn(r));
228
+ }
229
+ let bestIndex = 0;
230
+ for (let i = 1; i < scores.length; i++) if (scores[i] > scores[bestIndex]) bestIndex = i;
231
+ const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length;
232
+ return {
233
+ best: rollouts[bestIndex],
234
+ bestScore: scores[bestIndex],
235
+ scores,
236
+ meanScore,
237
+ bestIndex
238
+ };
239
+ }
240
+ /**
241
+ * Self-consistency / majority-vote test-time scaling. For tasks with a
242
+ * small categorical answer space (math problems, multiple choice).
243
+ */
223
244
  async function selfConsistency(opts) {
224
- if (opts.n <= 0) throw new ValidationError("selfConsistency: n must be > 0");
225
- const rollouts = [];
226
- const histogram = {};
227
- for (let i = 0; i < opts.n; i++) {
228
- const r = await opts.sample(i);
229
- rollouts.push(r);
230
- const key = opts.answerKey(r);
231
- histogram[key] = (histogram[key] ?? 0) + 1;
232
- }
233
- let answer = "";
234
- let max = -1;
235
- for (const [k, v] of Object.entries(histogram)) {
236
- if (v > max) {
237
- max = v;
238
- answer = k;
239
- }
240
- }
241
- const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0];
242
- return {
243
- answer,
244
- agreement: max / opts.n,
245
- histogram,
246
- representative,
247
- rollouts
248
- };
245
+ if (opts.n <= 0) throw new ValidationError("selfConsistency: n must be > 0");
246
+ const rollouts = [];
247
+ const histogram = {};
248
+ for (let i = 0; i < opts.n; i++) {
249
+ const r = await opts.sample(i);
250
+ rollouts.push(r);
251
+ const key = opts.answerKey(r);
252
+ histogram[key] = (histogram[key] ?? 0) + 1;
253
+ }
254
+ let answer = "";
255
+ let max = -1;
256
+ for (const [k, v] of Object.entries(histogram)) if (v > max) {
257
+ max = v;
258
+ answer = k;
259
+ }
260
+ const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0];
261
+ return {
262
+ answer,
263
+ agreement: max / opts.n,
264
+ histogram,
265
+ representative,
266
+ rollouts
267
+ };
249
268
  }
250
269
  function paretoFrontier(points) {
251
- const onFrontier = [];
252
- for (const p of points) {
253
- const dominated = points.some(
254
- (q) => q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score)
255
- );
256
- if (!dominated) onFrontier.push(p);
257
- }
258
- return onFrontier.sort((a, b) => a.cost - b.cost);
270
+ const onFrontier = [];
271
+ for (const p of points) if (!points.some((q) => q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score))) onFrontier.push(p);
272
+ return onFrontier.sort((a, b) => a.cost - b.cost);
259
273
  }
260
274
  function fitLogSlope(points) {
261
- const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)));
262
- const ys = points.map((p) => p.score);
263
- const n = xs.length;
264
- const mx = xs.reduce((s, x) => s + x, 0) / n;
265
- const my = ys.reduce((s, y) => s + y, 0) / n;
266
- let num = 0;
267
- let den = 0;
268
- for (let i = 0; i < n; i++) {
269
- num += (xs[i] - mx) * (ys[i] - my);
270
- den += (xs[i] - mx) ** 2;
271
- }
272
- return den === 0 ? 0 : num / den;
273
- }
274
-
275
- // src/rl/contamination.ts
275
+ const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)));
276
+ const ys = points.map((p) => p.score);
277
+ const n = xs.length;
278
+ const mx = xs.reduce((s, x) => s + x, 0) / n;
279
+ const my = ys.reduce((s, y) => s + y, 0) / n;
280
+ let num = 0;
281
+ let den = 0;
282
+ for (let i = 0; i < n; i++) {
283
+ num += (xs[i] - mx) * (ys[i] - my);
284
+ den += (xs[i] - mx) ** 2;
285
+ }
286
+ return den === 0 ? 0 : num / den;
287
+ }
288
+ //#endregion
289
+ //#region src/rl/contamination.ts
290
+ /**
291
+ * Contamination probe — held-out perturbation tests.
292
+ *
293
+ * The bug class: once a benchmark scenario set is published, models train
294
+ * on it, and your scores become invalid. SWE-Bench-Verified, GPQA, and
295
+ * MMLU-Pro all exist because their predecessors got contaminated within
296
+ * months. The right defense is to keep a held-out *perturbed* version of
297
+ * every scenario — same task, slightly different surface — and check
298
+ * whether scores diverge significantly. Genuine capability transfers; rote
299
+ * memorization doesn't.
300
+ *
301
+ * This module ships the probe contract:
302
+ *
303
+ * 1. A `ScenarioPerturbation` strategy type — function that produces a
304
+ * perturbed scenario from an original.
305
+ * 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs
306
+ * both halves and reports per-scenario score divergence + a global
307
+ * contamination verdict via paired Wilcoxon.
308
+ * 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,
309
+ * `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the
310
+ * task's structural difficulty while breaking surface memorization.
311
+ *
312
+ * The verdict is conservative: if the perturbed-vs-original score
313
+ * difference is statistically significant (BH-adjusted p < 0.05) AND
314
+ * the median drop is > 5 percentage points, we flag *contamination
315
+ * suspected*. False positives are possible (the perturbation might
316
+ * actually be harder); the default is to flag for review, not to
317
+ * autoreject.
318
+ */
276
319
  async function runContaminationProbe(input, opts = {}) {
277
- const fdr = opts.fdr ?? 0.05;
278
- const minMedianDrop = opts.minMedianDrop ?? 0.05;
279
- const floor = opts.scoreFloor ?? 0;
280
- if (!input.perturbed && !input.perturbation) {
281
- throw new ValidationError(
282
- "runContaminationProbe: must supply either `perturbed` or `perturbation`."
283
- );
284
- }
285
- const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
286
- if (perturbed.length !== input.originals.length) {
287
- throw new ValidationError(
288
- `runContaminationProbe: perturbed length ${perturbed.length} \u2260 originals ${input.originals.length}`
289
- );
290
- }
291
- const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
292
- const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
293
- const perScenario = input.originals.map((s, i) => ({
294
- scenarioId: input.scenarioId(s),
295
- originalScore: origScores[i],
296
- perturbedScore: pertScores[i],
297
- delta: pertScores[i] - origScores[i],
298
- qValue: NaN
299
- }));
300
- const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
301
- if (valid.length < 4) {
302
- return {
303
- perScenario,
304
- pairedTest: { w: 0, p: 1 },
305
- medianDelta: 0,
306
- meanDelta: 0,
307
- contaminationSuspected: false,
308
- reason: `insufficient valid scenarios (n=${valid.length}, need \u2265 4)`,
309
- n: valid.length
310
- };
311
- }
312
- const origValid = valid.map((p) => p.originalScore);
313
- const pertValid = valid.map((p) => p.perturbedScore);
314
- const pairedTest = wilcoxonSignedRank(origValid, pertValid);
315
- const deltas = valid.map((p) => p.delta);
316
- const sortedDeltas = [...deltas].sort((a, b) => a - b);
317
- const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
318
- const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
319
- const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)));
320
- const { qValues } = benjaminiHochberg(pseudoP, fdr);
321
- for (let i = 0; i < valid.length; i++) {
322
- const v = valid[i];
323
- const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
324
- if (idx >= 0) perScenario[idx].qValue = qValues[i];
325
- }
326
- const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
327
- const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
328
- return {
329
- perScenario,
330
- pairedTest,
331
- medianDelta: median,
332
- meanDelta: mean,
333
- contaminationSuspected,
334
- reason,
335
- n: valid.length
336
- };
337
- }
320
+ const fdr = opts.fdr ?? .05;
321
+ const minMedianDrop = opts.minMedianDrop ?? .05;
322
+ const floor = opts.scoreFloor ?? 0;
323
+ if (!input.perturbed && !input.perturbation) throw new ValidationError("runContaminationProbe: must supply either `perturbed` or `perturbation`.");
324
+ const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
325
+ if (perturbed.length !== input.originals.length) throw new ValidationError(`runContaminationProbe: perturbed length ${perturbed.length} ≠ originals ${input.originals.length}`);
326
+ const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
327
+ const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
328
+ const perScenario = input.originals.map((s, i) => ({
329
+ scenarioId: input.scenarioId(s),
330
+ originalScore: origScores[i],
331
+ perturbedScore: pertScores[i],
332
+ delta: pertScores[i] - origScores[i],
333
+ qValue: NaN
334
+ }));
335
+ const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
336
+ if (valid.length < 4) return {
337
+ perScenario,
338
+ pairedTest: {
339
+ w: 0,
340
+ p: 1
341
+ },
342
+ medianDelta: 0,
343
+ meanDelta: 0,
344
+ contaminationSuspected: false,
345
+ reason: `insufficient valid scenarios (n=${valid.length}, need ≥ 4)`,
346
+ n: valid.length
347
+ };
348
+ const pairedTest = wilcoxonSignedRank(valid.map((p) => p.originalScore), valid.map((p) => p.perturbedScore));
349
+ const deltas = valid.map((p) => p.delta);
350
+ const sortedDeltas = [...deltas].sort((a, b) => a - b);
351
+ const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
352
+ const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
353
+ const { qValues } = benjaminiHochberg(valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1))), fdr);
354
+ for (let i = 0; i < valid.length; i++) {
355
+ const v = valid[i];
356
+ const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
357
+ if (idx >= 0) perScenario[idx].qValue = qValues[i];
358
+ }
359
+ const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
360
+ return {
361
+ perScenario,
362
+ pairedTest,
363
+ medianDelta: median,
364
+ meanDelta: mean,
365
+ contaminationSuspected,
366
+ reason: contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`,
367
+ n: valid.length
368
+ };
369
+ }
370
+ /**
371
+ * Identifier-rename perturbation for code/text scenarios. Replaces every
372
+ * occurrence of the listed identifiers with synthesized aliases. Use when
373
+ * the scenario's structural difficulty is independent of variable names
374
+ * (e.g. SWE-Bench-style coding tasks).
375
+ */
338
376
  function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
339
- return {
340
- kind: "rename_variables",
341
- apply(scenario) {
342
- let prompt = scenario.prompt;
343
- identifiers.forEach((id, i) => {
344
- const replacement = rename(id, i);
345
- const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
346
- prompt = prompt.replace(re, replacement);
347
- });
348
- return { ...scenario, prompt };
349
- }
350
- };
351
- }
377
+ return {
378
+ kind: "rename_variables",
379
+ apply(scenario) {
380
+ let prompt = scenario.prompt;
381
+ identifiers.forEach((id, i) => {
382
+ const replacement = rename(id, i);
383
+ const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
384
+ prompt = prompt.replace(re, replacement);
385
+ });
386
+ return {
387
+ ...scenario,
388
+ prompt
389
+ };
390
+ }
391
+ };
392
+ }
393
+ /**
394
+ * Order-shuffle perturbation. Reshuffles a list-shaped section of the
395
+ * prompt (for QA scenarios that present options A/B/C/D — answer depends
396
+ * on the option labels, not order). Caller provides the section extractor.
397
+ */
352
398
  function shuffleOrder(shuffleSection, seed) {
353
- let s = seed >>> 0;
354
- const rng = () => {
355
- s = s + 1831565813 >>> 0;
356
- let t = s;
357
- t = Math.imul(t ^ t >>> 15, t | 1);
358
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
359
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
360
- };
361
- return {
362
- kind: "shuffle_order",
363
- apply(scenario) {
364
- const newPrompt = shuffleSection(scenario.prompt, rng);
365
- return { ...scenario, prompt: newPrompt };
366
- }
367
- };
368
- }
399
+ let s = seed >>> 0;
400
+ const rng = () => {
401
+ s = s + 1831565813 >>> 0;
402
+ let t = s;
403
+ t = Math.imul(t ^ t >>> 15, t | 1);
404
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
405
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
406
+ };
407
+ return {
408
+ kind: "shuffle_order",
409
+ apply(scenario) {
410
+ const newPrompt = shuffleSection(scenario.prompt, rng);
411
+ return {
412
+ ...scenario,
413
+ prompt: newPrompt
414
+ };
415
+ }
416
+ };
417
+ }
418
+ /**
419
+ * Inject-irrelevant-clause perturbation. Adds a benign sentence that
420
+ * shouldn't change the answer. Tests for "did the model just memorize
421
+ * the input string."
422
+ */
369
423
  function injectIrrelevantClause(clause, position = "prefix") {
370
- return {
371
- kind: "inject_irrelevant_clause",
372
- apply(scenario) {
373
- const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
374
- return { ...scenario, prompt };
375
- }
376
- };
424
+ return {
425
+ kind: "inject_irrelevant_clause",
426
+ apply(scenario) {
427
+ const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
428
+ return {
429
+ ...scenario,
430
+ prompt
431
+ };
432
+ }
433
+ };
377
434
  }
378
435
  function escapeRegex(s) {
379
- return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
380
- }
381
-
382
- // src/rl/corpus.ts
383
- import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
384
- import { dirname } from "path";
385
-
386
- // src/rl/rollout-input.ts
436
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
437
+ }
438
+ //#endregion
439
+ //#region src/rl/rollout-input.ts
440
+ /**
441
+ * Shared checks for trainer exports over canonical minted rollout lines.
442
+ *
443
+ * Exporters accept only `MintedRolloutLine[]`. Callers convert run records with
444
+ * `mintRolloutRows` before deriving preferences or trainer files.
445
+ */
446
+ /**
447
+ * The line's reward, with `null` meaning "no verdict exists" and never "scored
448
+ * zero".
449
+ *
450
+ * The wire contract represents an absent verdict directly as `reward: null`.
451
+ */
387
452
  function trainableLineReward(line) {
388
- assertRewardGate(line, "trainable reward");
389
- const { reward } = line.outcome;
390
- if (reward === null || !Number.isFinite(reward)) return null;
391
- return reward;
453
+ assertRewardGate(line, "trainable reward");
454
+ const { reward } = line.outcome;
455
+ if (reward === null || !Number.isFinite(reward)) return null;
456
+ return reward;
392
457
  }
458
+ /** The anti-Goodhart flag as it travels on the line. */
393
459
  function isLineRealnessGated(line) {
394
- return line.outcome.realness_gated === true;
460
+ return line.outcome.realness_gated === true;
395
461
  }
396
462
  function push(index, key, line) {
397
- const existing = index.get(key);
398
- if (existing === void 0) index.set(key, [line]);
399
- else existing.push(line);
463
+ const existing = index.get(key);
464
+ if (existing === void 0) index.set(key, [line]);
465
+ else existing.push(line);
400
466
  }
401
467
  function invocationIndex(lines) {
402
- const byRollout = /* @__PURE__ */ new Map();
403
- const byRun = /* @__PURE__ */ new Map();
404
- for (const line of lines) {
405
- push(byRollout, line.rollout_id, line);
406
- push(byRun, line.run_id, line);
407
- }
408
- return { byRollout, byRun };
409
- }
468
+ const byRollout = /* @__PURE__ */ new Map();
469
+ const byRun = /* @__PURE__ */ new Map();
470
+ for (const line of lines) {
471
+ push(byRollout, line.rollout_id, line);
472
+ push(byRun, line.run_id, line);
473
+ }
474
+ return {
475
+ byRollout,
476
+ byRun
477
+ };
478
+ }
479
+ /**
480
+ * Resolve one referenced id to exactly ONE invocation, or refuse to guess.
481
+ *
482
+ * The previous implementation was `new Map(lines.map((l) => [l.run_id, l]))`,
483
+ * which is LAST-WINS: with a gated supervisor node and an ungated worker sharing
484
+ * a `run_id`, the answer depended on which one appeared later in the array, so
485
+ * `[gatedRoot, worker, rival]` emitted the gamed trajectory as `chosen` and
486
+ * simply reordering the same input suppressed it. An order-dependent security
487
+ * property passes every test whose fixture happens to be ordered favourably,
488
+ * which is the worst possible failure mode for a gate.
489
+ *
490
+ * The rule that removes order from the answer: an id resolves only when it names
491
+ * one invocation. `rollout_id` is tried first because it IS the invocation id;
492
+ * `run_id` is accepted only when the run holds a single invocation, and a
493
+ * cross-index disagreement (an id that is one line's `rollout_id` and a
494
+ * different line's `run_id`) is ambiguous rather than silently preferring
495
+ * either.
496
+ */
410
497
  function resolveInvocation(index, id) {
411
- const rollouts = index.byRollout.get(id) ?? [];
412
- const runs = index.byRun.get(id) ?? [];
413
- if (rollouts.length > 1) return { kind: "ambiguous", count: rollouts.length };
414
- const exact = rollouts[0];
415
- if (exact !== void 0) {
416
- if (runs.some((line) => line !== exact)) {
417
- return { kind: "ambiguous", count: 1 + runs.filter((line) => line !== exact).length };
418
- }
419
- return { kind: "resolved", line: exact };
420
- }
421
- if (runs.length > 1) return { kind: "ambiguous", count: runs.length };
422
- const only = runs[0];
423
- return only === void 0 ? { kind: "missing" } : { kind: "resolved", line: only };
424
- }
498
+ const rollouts = index.byRollout.get(id) ?? [];
499
+ const runs = index.byRun.get(id) ?? [];
500
+ if (rollouts.length > 1) return {
501
+ kind: "ambiguous",
502
+ count: rollouts.length
503
+ };
504
+ const exact = rollouts[0];
505
+ if (exact !== void 0) {
506
+ if (runs.some((line) => line !== exact)) return {
507
+ kind: "ambiguous",
508
+ count: 1 + runs.filter((line) => line !== exact).length
509
+ };
510
+ return {
511
+ kind: "resolved",
512
+ line: exact
513
+ };
514
+ }
515
+ if (runs.length > 1) return {
516
+ kind: "ambiguous",
517
+ count: runs.length
518
+ };
519
+ const only = runs[0];
520
+ return only === void 0 ? { kind: "missing" } : {
521
+ kind: "resolved",
522
+ line: only
523
+ };
524
+ }
525
+ /**
526
+ * THE admission rule for every exporter whose input is line-less — one
527
+ * implementation, because two siblings over the same input class with different
528
+ * gating is the defect being eliminated, and it has now happened twice
529
+ * (`toPrmRows` hardened while `toDpoRows` was left open; `toGrpoRows`'
530
+ * `rewardOf` gated while `extractPreferences`' identically-named hook was not).
531
+ *
532
+ * Fail-closed in five steps:
533
+ * 1. No context at all → throw. A two-argument call used to be accepted and
534
+ * produced rows with no gate applied whatsoever.
535
+ * 2. A referenced id with NO line → throw. Its gate status is unknown, and
536
+ * unknown is not clean. Thrown rather than dropped because it means the
537
+ * caller did not supply the context it was asked for, which is a defect in
538
+ * the call, not in the data.
539
+ * 3. A referenced id naming MORE THAN ONE invocation → DROP the item and count
540
+ * it. Dropped rather than thrown because, unlike (2), this is ordinary data
541
+ * — a supervision episode legitimately holds a supervisor invocation and
542
+ * several workers under one `run_id` — and throwing would make these
543
+ * exporters unusable on any supervisor corpus, whose only workaround is for
544
+ * the caller to hand-filter `context.lines` down to one line per run. That
545
+ * workaround IS the leak, performed by hand. The count is surfaced by
546
+ * `admitUngatedByInvocation` so the drop is never silent.
547
+ * 4. Every resolved line goes through `assertRewardGate`, so the line-less
548
+ * exporters compose the same check list as the waist exporters instead of
549
+ * relying on `realness_gated` alone (which is one of three checks).
550
+ * 5. Either side realness-gated → DROP the item. Dropped rather than zeroed
551
+ * because these shapes have no honest zero: a preference pair is a
552
+ * statement that one trajectory is better than another, and a gamed
553
+ * trajectory belongs on neither side of it — as the chosen one it teaches
554
+ * the gaming move outright, and as the rejected one it still ships the
555
+ * gaming trajectory's text into the training file as a contrast example
556
+ * nobody asked for.
557
+ *
558
+ * `inspect` is the per-exporter extra check (PRM's trajectory-completeness
559
+ * rules). It runs on every resolved line before any item is admitted, so the
560
+ * whole batch fails before a single row is built.
561
+ *
562
+ * Pure: it reports what it dropped and prints nothing.
563
+ */
425
564
  function auditInvocationAdmission(items, idsOf, context, requirement, inspect) {
426
- if (context === void 0 || context === null) {
427
- throw new Error(
428
- `${requirement.exporter}: a ${requirement.contextType} is required \u2014 ${requirement.because} Pass \`{ lines: (await mintRolloutRows(...)).rows }\`.`
429
- );
430
- }
431
- const index = invocationIndex(context.lines);
432
- const audit = {
433
- admitted: [],
434
- gatedDrops: 0,
435
- ambiguousDrops: 0,
436
- ambiguous: []
437
- };
438
- for (const item of items) {
439
- const lines = [];
440
- let ambiguous = false;
441
- for (const id of idsOf(item)) {
442
- const resolution = resolveInvocation(index, id);
443
- if (resolution.kind === "missing") {
444
- throw new Error(
445
- `${requirement.exporter}: no rollout line supplied for run ${id} \u2014 its realness gate and capture quality are unknown`
446
- );
447
- }
448
- if (resolution.kind === "ambiguous") {
449
- ambiguous = true;
450
- if (!audit.ambiguous.some((entry) => entry.id === id)) {
451
- audit.ambiguous.push({ id, invocations: resolution.count });
452
- }
453
- continue;
454
- }
455
- lines.push(resolution.line);
456
- }
457
- for (const line of lines) {
458
- assertRewardGate(line, requirement.exporter);
459
- inspect?.(line);
460
- }
461
- if (ambiguous) {
462
- audit.ambiguousDrops++;
463
- continue;
464
- }
465
- if (lines.some(isLineRealnessGated)) {
466
- audit.gatedDrops++;
467
- continue;
468
- }
469
- audit.admitted.push(item);
470
- }
471
- return audit;
472
- }
565
+ if (context === void 0 || context === null) throw new Error(`${requirement.exporter}: a ${requirement.contextType} is required — ${requirement.because} Pass \`{ lines: (await mintRolloutRows(...)).rows }\`.`);
566
+ const index = invocationIndex(context.lines);
567
+ const audit = {
568
+ admitted: [],
569
+ gatedDrops: 0,
570
+ ambiguousDrops: 0,
571
+ ambiguous: []
572
+ };
573
+ for (const item of items) {
574
+ const lines = [];
575
+ let ambiguous = false;
576
+ for (const id of idsOf(item)) {
577
+ const resolution = resolveInvocation(index, id);
578
+ if (resolution.kind === "missing") throw new Error(`${requirement.exporter}: no rollout line supplied for run ${id} — its realness gate and capture quality are unknown`);
579
+ if (resolution.kind === "ambiguous") {
580
+ ambiguous = true;
581
+ if (!audit.ambiguous.some((entry) => entry.id === id)) audit.ambiguous.push({
582
+ id,
583
+ invocations: resolution.count
584
+ });
585
+ continue;
586
+ }
587
+ lines.push(resolution.line);
588
+ }
589
+ for (const line of lines) {
590
+ assertRewardGate(line, requirement.exporter);
591
+ inspect?.(line);
592
+ }
593
+ if (ambiguous) {
594
+ audit.ambiguousDrops++;
595
+ continue;
596
+ }
597
+ if (lines.some(isLineRealnessGated)) {
598
+ audit.gatedDrops++;
599
+ continue;
600
+ }
601
+ audit.admitted.push(item);
602
+ }
603
+ return audit;
604
+ }
605
+ /**
606
+ * `auditInvocationAdmission` for the exporters, which return rows and have
607
+ * nowhere to put a count.
608
+ *
609
+ * The ambiguous drops are announced rather than swallowed: a caller who asked
610
+ * for N pairs and silently received N-k has no way to notice that a chunk of
611
+ * their preference data quietly evaporated, and "the training set got smaller
612
+ * for a reason nobody printed" is the same class of invisible failure as the
613
+ * gate that never ran. A gated drop is NOT announced — that one is the gate
614
+ * doing exactly its job, on the population the caller already knows is flagged.
615
+ */
473
616
  function admitUngatedByInvocation(items, idsOf, context, requirement, inspect) {
474
- const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect);
475
- if (audit.ambiguousDrops > 0) {
476
- const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(", ");
477
- console.warn(
478
- `[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more than one invocation in the supplied lines, so the realness gate cannot be read for the invocation the artifact meant. Reference the \`rollout_id\` instead of the \`run_id\`, or supply a context holding one invocation per run.`
479
- );
480
- }
481
- return audit.admitted;
482
- }
483
-
484
- // src/rl/exporters.ts
485
- var DPO_CONTEXT_REQUIREMENT = {
486
- exporter: "DPO export",
487
- contextType: "DpoLineContext",
488
- because: "a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair \u2014 which is DPO trained to PREFER the gaming trajectory."
617
+ const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect);
618
+ if (audit.ambiguousDrops > 0) {
619
+ const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(", ");
620
+ console.warn(`[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more than one invocation in the supplied lines, so the realness gate cannot be read for the invocation the artifact meant. Reference the \`rollout_id\` instead of the \`run_id\`, or supply a context holding one invocation per run.`);
621
+ }
622
+ return audit.admitted;
623
+ }
624
+ //#endregion
625
+ //#region src/rl/exporters.ts
626
+ /**
627
+ * Trainer-format exporters.
628
+ *
629
+ * agent-eval produces canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`,
630
+ * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
631
+ * different shapes Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
632
+ * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
633
+ * JSONL conventions. Rather than ship N adapters, this module ships the
634
+ * canonical formats most production pipelines accept and ergonomic helpers
635
+ * for the rest.
636
+ *
637
+ * Shapes:
638
+ * - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed
639
+ * by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.
640
+ * - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.
641
+ * Consumed by prime-rl GRPO, verl, OpenRLHF.
642
+ * - **SFT** — `{messages[]}` JSONL with chosen completion as the final
643
+ * assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,
644
+ * Anthropic finetuning.
645
+ * - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.
646
+ * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
647
+ *
648
+ * Why ship this in agent-eval rather than a separate adapter package: the
649
+ * canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`, etc.) are
650
+ * agent-eval's contract; without first-party exporters consumers reverse-
651
+ * engineer the mapping every release. The exporters codify it.
652
+ *
653
+ * The exporters take callbacks for any field that isn't on the canonical
654
+ * artifact (specifically: prompt + completion text, since the package
655
+ * stores only their hashes by design — full text is the consumer's
656
+ * trace store / raw event log).
657
+ *
658
+ * Every exporter that produces a training row accepts canonical minted rollout
659
+ * lines. Convert run records once with `mintRolloutRows`; downstream transforms
660
+ * then share one reward, split, and authenticity contract.
661
+ */
662
+ const DPO_CONTEXT_REQUIREMENT = {
663
+ exporter: "DPO export",
664
+ contextType: "DpoLineContext",
665
+ because: "a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair — which is DPO trained to PREFER the gaming trajectory."
489
666
  };
667
+ /**
668
+ * Convert preference triples to TRL-compatible DPO rows. The shape
669
+ * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
670
+ * entry; every major DPO trainer accepts it.
671
+ *
672
+ * `context` is REQUIRED, and for the same reason it is required on the sibling
673
+ * `toPrmRows`: a triple is a line-less artifact. It names two run ids and a
674
+ * margin, and nothing on it says whether either run was flagged as gamed —
675
+ * so a two-argument call applied NO gate at all and emitted the row verbatim,
676
+ * reachable straight through the published bundle builder
677
+ * (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).
678
+ * Triples whose chosen or rejected side is realness-gated are dropped; a triple
679
+ * naming a run with no supplied line is refused. See `admitUngatedByInvocation` for
680
+ * why dropping, not zeroing, is the right disposition for a preference pair.
681
+ */
490
682
  async function toDpoRows(triples, lookups, context) {
491
- const admitted = admitUngatedByInvocation(
492
- triples,
493
- (t) => [t.chosenRunId, t.rejectedRunId],
494
- context,
495
- DPO_CONTEXT_REQUIREMENT
496
- );
497
- const out = [];
498
- for (const t of admitted) {
499
- const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
500
- Promise.resolve(lookups.promptOf(t.chosenRunId)),
501
- Promise.resolve(lookups.promptOf(t.rejectedRunId)),
502
- Promise.resolve(lookups.completionOf(t.chosenRunId)),
503
- Promise.resolve(lookups.completionOf(t.rejectedRunId))
504
- ]);
505
- if (chosenPrompt !== rejectedPrompt) {
506
- throw new Error(
507
- `toDpoRows: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`
508
- );
509
- }
510
- out.push({
511
- prompt: chosenPrompt,
512
- chosen,
513
- rejected,
514
- margin: t.marginScore,
515
- meta: {
516
- scenarioId: t.scenarioId,
517
- chosenVariantId: t.chosenVariantId,
518
- rejectedVariantId: t.rejectedVariantId,
519
- chosenRunId: t.chosenRunId,
520
- rejectedRunId: t.rejectedRunId,
521
- chosenModel: t.meta.chosenModel,
522
- rejectedModel: t.meta.rejectedModel
523
- }
524
- });
525
- }
526
- return out;
527
- }
683
+ const admitted = admitUngatedByInvocation(triples, (t) => [t.chosenRunId, t.rejectedRunId], context, DPO_CONTEXT_REQUIREMENT);
684
+ const out = [];
685
+ for (const t of admitted) {
686
+ const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
687
+ Promise.resolve(lookups.promptOf(t.chosenRunId)),
688
+ Promise.resolve(lookups.promptOf(t.rejectedRunId)),
689
+ Promise.resolve(lookups.completionOf(t.chosenRunId)),
690
+ Promise.resolve(lookups.completionOf(t.rejectedRunId))
691
+ ]);
692
+ if (chosenPrompt !== rejectedPrompt) throw new Error(`toDpoRows: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`);
693
+ out.push({
694
+ prompt: chosenPrompt,
695
+ chosen,
696
+ rejected,
697
+ margin: t.marginScore,
698
+ meta: {
699
+ scenarioId: t.scenarioId,
700
+ chosenVariantId: t.chosenVariantId,
701
+ rejectedVariantId: t.rejectedVariantId,
702
+ chosenRunId: t.chosenRunId,
703
+ rejectedRunId: t.rejectedRunId,
704
+ chosenModel: t.meta.chosenModel,
705
+ rejectedModel: t.meta.rejectedModel
706
+ }
707
+ });
708
+ }
709
+ return out;
710
+ }
711
+ /** Serialize DPO rows as JSONL. One line per row. */
528
712
  function toDpoJsonl(rows) {
529
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
530
- }
713
+ return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
714
+ }
715
+ /**
716
+ * Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —
717
+ * one row per scenario, with one completion per rollout on that scenario.
718
+ * A scenario with fewer than two rewarded completions emits no row because a
719
+ * group of one has no relative baseline.
720
+ *
721
+ * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
722
+ * within a group of completions for the same prompt; this is the
723
+ * canonical input format. That relative baseline is exactly why the gate has
724
+ * to hold here: one gamed sibling exporting at full reward shifts the advantage
725
+ * of every honest run beside it.
726
+ *
727
+ * On the line path a realness-gated line stays in its group at reward 0 rather
728
+ * than being dropped. 0 is the honest label for a faked success and is usable
729
+ * signal; removing the line would also move the group's baseline, just in the
730
+ * other direction. (SFT differs — see `toSftRows`.)
731
+ */
531
732
  async function toGrpoRows(lines, lookups) {
532
- return grpoRowsFromLines(lines, lookups);
733
+ return grpoRowsFromLines(lines, lookups);
533
734
  }
534
735
  async function grpoRowsFromLines(lines, lookups) {
535
- const grouped = /* @__PURE__ */ new Map();
536
- for (const line of lines) {
537
- if (!isSelectedSplit(line, lookups)) continue;
538
- const arr = grouped.get(line.task.instance_id) ?? [];
539
- arr.push(line);
540
- grouped.set(line.task.instance_id, arr);
541
- }
542
- const rows = [];
543
- for (const [scenarioId, group] of grouped.entries()) {
544
- if (group.length === 0) continue;
545
- const scored = [];
546
- for (const line of group) {
547
- const reward = trainableLineReward(line);
548
- if (reward === null) continue;
549
- scored.push({ line, reward });
550
- }
551
- if (scored.length < 2) continue;
552
- const prompts = await Promise.all(
553
- scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id)))
554
- );
555
- const prompt = prompts[0];
556
- if (prompts.some((value) => value !== prompt)) {
557
- throw new Error(
558
- `toGrpoRows: scenario "${scenarioId}" resolves to different prompt text within one group`
559
- );
560
- }
561
- const completions = await Promise.all(
562
- scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id)))
563
- );
564
- const rewards = scored.map(({ reward }) => reward);
565
- const runIds = scored.map(({ line }) => line.run_id);
566
- rows.push({
567
- prompt,
568
- completions,
569
- rewards,
570
- runIds,
571
- meta: {
572
- scenarioId,
573
- n: completions.length,
574
- meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
575
- }
576
- });
577
- }
578
- return rows;
736
+ const grouped = /* @__PURE__ */ new Map();
737
+ for (const line of lines) {
738
+ if (!isSelectedSplit(line, lookups)) continue;
739
+ const arr = grouped.get(line.task.instance_id) ?? [];
740
+ arr.push(line);
741
+ grouped.set(line.task.instance_id, arr);
742
+ }
743
+ const rows = [];
744
+ for (const [scenarioId, group] of grouped.entries()) {
745
+ if (group.length === 0) continue;
746
+ const scored = [];
747
+ for (const line of group) {
748
+ const reward = trainableLineReward(line);
749
+ if (reward === null) continue;
750
+ scored.push({
751
+ line,
752
+ reward
753
+ });
754
+ }
755
+ if (scored.length < 2) continue;
756
+ const prompts = await Promise.all(scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id))));
757
+ const prompt = prompts[0];
758
+ if (prompts.some((value) => value !== prompt)) throw new Error(`toGrpoRows: scenario "${scenarioId}" resolves to different prompt text within one group`);
759
+ const completions = await Promise.all(scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id))));
760
+ const rewards = scored.map(({ reward }) => reward);
761
+ const runIds = scored.map(({ line }) => line.run_id);
762
+ rows.push({
763
+ prompt,
764
+ completions,
765
+ rewards,
766
+ runIds,
767
+ meta: {
768
+ scenarioId,
769
+ n: completions.length,
770
+ meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
771
+ }
772
+ });
773
+ }
774
+ return rows;
579
775
  }
580
776
  function toGrpoJsonl(rows) {
581
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
582
- }
777
+ return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
778
+ }
779
+ /**
780
+ * Convert rollout lines into Hugging Face / OpenAI / Anthropic-style
781
+ * conversational SFT rows. By default every qualifying line becomes one row;
782
+ * pass `include` to filter further (e.g., keep only `reward >= 0.8` for
783
+ * rejection-sampling SFT).
784
+ *
785
+ * Realness-gated lines are dropped outright, not zeroed. SFT is imitation
786
+ * learning: unlike GRPO, where a 0 reward teaches "this trajectory was bad",
787
+ * every row here is a target to copy, so a gamed trajectory must not be in the
788
+ * file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.
789
+ *
790
+ * The exporter is fail-closed on the split, same rule as
791
+ * `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by
792
+ * default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and
793
+ * `canary` never pass the default rule. A non-training bundle that wants an
794
+ * explicit slice (e.g. a holdout-only eval bundle) names it with
795
+ * `splitFilter: ['holdout']` — explicit selection replaces the default rule.
796
+ */
583
797
  async function toSftRows(lines, lookups) {
584
- return sftRowsFromLines(lines, lookups);
798
+ return sftRowsFromLines(lines, lookups);
585
799
  }
586
800
  async function sftRowsFromLines(lines, lookups) {
587
- const include = lookups.include ?? (() => true);
588
- const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0;
589
- if (!Number.isFinite(minimumQualityExclusive)) {
590
- throw new Error("minimumQualityExclusive must be finite");
591
- }
592
- const rows = [];
593
- for (const line of lines) {
594
- assertRewardGate(line, "SFT export");
595
- if (isLineRealnessGated(line)) continue;
596
- if (!isSelectedSplit(line, lookups)) continue;
597
- const score = trainableLineReward(line);
598
- if (score === null || score <= minimumQualityExclusive) continue;
599
- if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {
600
- continue;
601
- }
602
- if (!include(line)) continue;
603
- const system = lookups.systemOf?.(line);
604
- const [prompt, completion] = await Promise.all([
605
- Promise.resolve(lookups.promptOf(line.run_id)),
606
- Promise.resolve(lookups.completionOf(line.run_id))
607
- ]);
608
- const messages = [];
609
- if (system) messages.push({ role: "system", content: system });
610
- messages.push({ role: "user", content: prompt });
611
- messages.push({ role: "assistant", content: completion });
612
- rows.push({
613
- messages,
614
- meta: {
615
- runId: line.run_id,
616
- candidateId: line.candidate_id ?? null,
617
- scenarioId: line.task.instance_id,
618
- score,
619
- model: line.policy.model
620
- }
621
- });
622
- }
623
- return rows;
801
+ const include = lookups.include ?? (() => true);
802
+ const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0;
803
+ if (!Number.isFinite(minimumQualityExclusive)) throw new Error("minimumQualityExclusive must be finite");
804
+ const rows = [];
805
+ for (const line of lines) {
806
+ assertRewardGate(line, "SFT export");
807
+ if (isLineRealnessGated(line)) continue;
808
+ if (!isSelectedSplit(line, lookups)) continue;
809
+ const score = trainableLineReward(line);
810
+ if (score === null || score <= minimumQualityExclusive) continue;
811
+ if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) continue;
812
+ if (!include(line)) continue;
813
+ const system = lookups.systemOf?.(line);
814
+ const [prompt, completion] = await Promise.all([Promise.resolve(lookups.promptOf(line.run_id)), Promise.resolve(lookups.completionOf(line.run_id))]);
815
+ const messages = [];
816
+ if (system) messages.push({
817
+ role: "system",
818
+ content: system
819
+ });
820
+ messages.push({
821
+ role: "user",
822
+ content: prompt
823
+ });
824
+ messages.push({
825
+ role: "assistant",
826
+ content: completion
827
+ });
828
+ rows.push({
829
+ messages,
830
+ meta: {
831
+ runId: line.run_id,
832
+ candidateId: line.candidate_id ?? null,
833
+ scenarioId: line.task.instance_id,
834
+ score,
835
+ model: line.policy.model
836
+ }
837
+ });
838
+ }
839
+ return rows;
624
840
  }
625
841
  function toSftJsonl(rows) {
626
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
627
- }
842
+ return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
843
+ }
844
+ /**
845
+ * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
846
+ * callback resolves span text from the consumer's trace store.
847
+ *
848
+ * Every referenced run is checked against its minted line before any row is
849
+ * emitted, and the export FAILS LOUD on a trajectory that was never fully
850
+ * captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected
851
+ * side is realness-gated are dropped instead: a capture defect is the caller's
852
+ * mint configuration and must be fixed, whereas a gamed run is exactly the
853
+ * condition the gate exists to filter.
854
+ *
855
+ * `context` is REQUIRED. A two-argument call used to be accepted and produced
856
+ * rows with no gate applied at all — a `PrmTrainingTriple` carries a bare
857
+ * `chosenReward` number and nothing that says which run it came from is honest,
858
+ * so with no lines this exporter has no way to learn that its chosen step is a
859
+ * step from a run that faked its success. It now throws: fail closed, because
860
+ * the alternative is a process-reward model taught to prefer the gaming move at
861
+ * the exact step the gaming happened.
862
+ */
628
863
  async function toPrmRows(triples, lookups, context) {
629
- const admitted = admitPrmTriples(triples, context);
630
- const rows = [];
631
- for (const t of admitted) {
632
- const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
633
- const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
634
- const prefixStepText = [];
635
- for (const spanId of prefixSpanIds) {
636
- prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)));
637
- }
638
- const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId));
639
- const rejectedStep = await Promise.resolve(
640
- lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId)
641
- );
642
- rows.push({
643
- prompt,
644
- prefixSpanIds,
645
- prefixStepText,
646
- chosenStep,
647
- rejectedStep,
648
- chosenReward: t.chosenReward,
649
- rejectedReward: t.rejectedReward,
650
- marginScore: t.marginScore,
651
- meta: {
652
- prefixRunId: t.prefixRunId,
653
- rejectedRunId: t.rejectedRunId,
654
- prefixStepIndex: t.prefixStepIndex
655
- }
656
- });
657
- }
658
- return rows;
659
- }
864
+ const admitted = admitPrmTriples(triples, context);
865
+ const rows = [];
866
+ for (const t of admitted) {
867
+ const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
868
+ const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
869
+ const prefixStepText = [];
870
+ for (const spanId of prefixSpanIds) prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)));
871
+ const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId));
872
+ const rejectedStep = await Promise.resolve(lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId));
873
+ rows.push({
874
+ prompt,
875
+ prefixSpanIds,
876
+ prefixStepText,
877
+ chosenStep,
878
+ rejectedStep,
879
+ chosenReward: t.chosenReward,
880
+ rejectedReward: t.rejectedReward,
881
+ marginScore: t.marginScore,
882
+ meta: {
883
+ prefixRunId: t.prefixRunId,
884
+ rejectedRunId: t.rejectedRunId,
885
+ prefixStepIndex: t.prefixStepIndex
886
+ }
887
+ });
888
+ }
889
+ return rows;
890
+ }
891
+ /**
892
+ * Refuse to build a process-reward row from a trajectory we do not fully have.
893
+ *
894
+ * PRM training assigns credit step by step, so a missing or silently shortened
895
+ * step list is not degraded data — it is data about a trajectory that never
896
+ * existed. Every condition below throws rather than filters, because each one
897
+ * means the CALLER's capture or mint configuration is wrong.
898
+ */
660
899
  function assertPrmTrainableLine(line, mintedWithMaxSteps) {
661
- const id = line.rollout_id;
662
- if (line.provenance.gap !== void 0) {
663
- throw new Error(
664
- `PRM export: rollout ${id} is a gap line (${line.provenance.gap}) \u2014 refusing to build a process-reward row from a trajectory that was never captured`
665
- );
666
- }
667
- if (line.steps === void 0 || line.steps.length === 0) {
668
- throw new Error(
669
- `PRM export: rollout ${id} carries no steps \u2014 refusing to build a process-reward row with no trajectory`
670
- );
671
- }
672
- if (line.outcome.is_truncated) {
673
- throw new Error(
674
- `PRM export: rollout ${id} is marked truncated \u2014 refusing to assign step-level credit over a partial trajectory`
675
- );
676
- }
677
- if (mintedWithMaxSteps !== void 0 && line.steps.length >= mintedWithMaxSteps) {
678
- throw new Error(
679
- `PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} \u2014 its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`
680
- );
681
- }
682
- }
683
- var PRM_CONTEXT_REQUIREMENT = {
684
- exporter: "PRM export",
685
- contextType: "PrmLineContext",
686
- because: "without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one."
900
+ const id = line.rollout_id;
901
+ if (line.provenance.gap !== void 0) throw new Error(`PRM export: rollout ${id} is a gap line (${line.provenance.gap}) — refusing to build a process-reward row from a trajectory that was never captured`);
902
+ if (line.steps === void 0 || line.steps.length === 0) throw new Error(`PRM export: rollout ${id} carries no steps — refusing to build a process-reward row with no trajectory`);
903
+ if (line.outcome.is_truncated) throw new Error(`PRM export: rollout ${id} is marked truncated refusing to assign step-level credit over a partial trajectory`);
904
+ if (mintedWithMaxSteps !== void 0 && line.steps.length >= mintedWithMaxSteps) throw new Error(`PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} — its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`);
905
+ }
906
+ const PRM_CONTEXT_REQUIREMENT = {
907
+ exporter: "PRM export",
908
+ contextType: "PrmLineContext",
909
+ because: "without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one."
687
910
  };
911
+ /**
912
+ * Validate every referenced line up front (fail loud, before a single row is
913
+ * written) and then drop the triples whose evidence is realness-gated.
914
+ *
915
+ * The gate half is `admitUngatedByInvocation`, shared with `toDpoRows` and
916
+ * `stepRewardsToJsonl`; only the trajectory-completeness rules are PRM's own.
917
+ */
688
918
  function admitPrmTriples(triples, context) {
689
- return admitUngatedByInvocation(
690
- triples,
691
- (t) => [t.prefixRunId, t.rejectedRunId],
692
- context,
693
- PRM_CONTEXT_REQUIREMENT,
694
- (line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps)
695
- );
919
+ return admitUngatedByInvocation(triples, (t) => [t.prefixRunId, t.rejectedRunId], context, PRM_CONTEXT_REQUIREMENT, (line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps));
696
920
  }
697
921
  function toPrmJsonl(rows) {
698
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
922
+ return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
699
923
  }
700
- var STEP_REWARD_CONTEXT_REQUIREMENT = {
701
- exporter: "step-reward export",
702
- contextType: "RolloutLineContext",
703
- because: "a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success \u2014 so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere."
924
+ const STEP_REWARD_CONTEXT_REQUIREMENT = {
925
+ exporter: "step-reward export",
926
+ contextType: "RolloutLineContext",
927
+ because: "a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere."
704
928
  };
929
+ /**
930
+ * Step-level reward rows as JSONL.
931
+ *
932
+ * `context` is REQUIRED for the same reason it is on `toDpoRows` and
933
+ * `toPrmRows`: this is a line-less input carrying a reward number. Steps
934
+ * belonging to a realness-gated run are dropped rather than zeroed — a
935
+ * per-step reward of 0 across a whole trajectory is a claim that every step was
936
+ * bad, which is a different (and false) statement from "this run's success was
937
+ * fabricated, so its step-level credit assignment is meaningless".
938
+ */
705
939
  function stepRewardsToJsonl(stepRewards, context) {
706
- const admitted = admitUngatedByInvocation(
707
- stepRewards,
708
- (s) => [s.runId],
709
- context,
710
- STEP_REWARD_CONTEXT_REQUIREMENT
711
- );
712
- const rows = admitted.map((s) => ({
713
- runId: s.runId,
714
- spanId: s.spanId,
715
- stepIndex: s.stepIndex,
716
- reward: s.reward,
717
- determinism: s.determinism,
718
- weight: s.weight ?? 1
719
- }));
720
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
940
+ const rows = admitUngatedByInvocation(stepRewards, (s) => [s.runId], context, STEP_REWARD_CONTEXT_REQUIREMENT).map((s) => ({
941
+ runId: s.runId,
942
+ spanId: s.spanId,
943
+ stepIndex: s.stepIndex,
944
+ reward: s.reward,
945
+ determinism: s.determinism,
946
+ weight: s.weight ?? 1
947
+ }));
948
+ return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
721
949
  }
722
950
  function isSelectedSplit(line, options) {
723
- if (options.splitFilter !== void 0) return options.splitFilter.includes(line.task.split);
724
- return isSplitEligible(line, options);
725
- }
726
-
727
- // src/rl/dataset.ts
728
- var DATASET_FORMATS = ["grpo", "sft", "dpo"];
729
- var DATASET_FORMAT_SET = new Set(DATASET_FORMATS);
951
+ if (options.splitFilter !== void 0) return options.splitFilter.includes(line.task.split);
952
+ return isSplitEligible(line, options);
953
+ }
954
+ //#endregion
955
+ //#region src/rl/dataset.ts
956
+ const DATASET_FORMAT_SET = /* @__PURE__ */ new Set([
957
+ "grpo",
958
+ "sft",
959
+ "dpo"
960
+ ]);
730
961
  function validateDatasetFormats(value) {
731
- if (!Array.isArray(value) || value.length === 0) {
732
- throw new Error("buildRlDataset: formats must contain at least one of: grpo, sft, dpo");
733
- }
734
- const formats = [];
735
- const seen = /* @__PURE__ */ new Set();
736
- for (const format of value) {
737
- if (!DATASET_FORMAT_SET.has(format)) {
738
- throw new Error(
739
- `buildRlDataset: unsupported format ${JSON.stringify(format)}; expected exactly one of: grpo, sft, dpo`
740
- );
741
- }
742
- const datasetFormat = format;
743
- if (seen.has(datasetFormat)) {
744
- throw new Error(
745
- `buildRlDataset: duplicate format ${JSON.stringify(datasetFormat)}; each format may be requested once`
746
- );
747
- }
748
- seen.add(datasetFormat);
749
- formats.push(datasetFormat);
750
- }
751
- return formats;
962
+ if (!Array.isArray(value) || value.length === 0) throw new Error("buildRlDataset: formats must contain at least one of: grpo, sft, dpo");
963
+ const formats = [];
964
+ const seen = /* @__PURE__ */ new Set();
965
+ for (const format of value) {
966
+ if (!DATASET_FORMAT_SET.has(format)) throw new Error(`buildRlDataset: unsupported format ${JSON.stringify(format)}; expected exactly one of: grpo, sft, dpo`);
967
+ const datasetFormat = format;
968
+ if (seen.has(datasetFormat)) throw new Error(`buildRlDataset: duplicate format ${JSON.stringify(datasetFormat)}; each format may be requested once`);
969
+ seen.add(datasetFormat);
970
+ formats.push(datasetFormat);
971
+ }
972
+ return formats;
752
973
  }
753
974
  function distinct(xs) {
754
- return [...new Set(xs.filter((x) => typeof x === "string" && x.length > 0))].sort();
975
+ return [...new Set(xs.filter((x) => typeof x === "string" && x.length > 0))].sort();
755
976
  }
756
977
  function computeRewardStats(values) {
757
- if (values.length === 0) {
758
- return { n: 0, mean: null, median: null, min: null, max: null, std: null };
759
- }
760
- const sorted = [...values].sort((a, b) => a - b);
761
- const n = sorted.length;
762
- const mean = sorted.reduce((s, x) => s + x, 0) / n;
763
- const mid = Math.floor(n / 2);
764
- const median = n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
765
- const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
766
- return { n, mean, median, min: sorted[0], max: sorted[n - 1], std: Math.sqrt(variance) };
978
+ if (values.length === 0) return {
979
+ n: 0,
980
+ mean: null,
981
+ median: null,
982
+ min: null,
983
+ max: null,
984
+ std: null
985
+ };
986
+ const sorted = [...values].sort((a, b) => a - b);
987
+ const n = sorted.length;
988
+ const mean = sorted.reduce((s, x) => s + x, 0) / n;
989
+ const mid = Math.floor(n / 2);
990
+ const median = n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
991
+ const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
992
+ return {
993
+ n,
994
+ mean,
995
+ median,
996
+ min: sorted[0],
997
+ max: sorted[n - 1],
998
+ std: Math.sqrt(variance)
999
+ };
767
1000
  }
768
1001
  function computeStatsFromLines(lines) {
769
- const splits = { search: 0, dev: 0, holdout: 0, canary: 0 };
770
- let inTok = 0;
771
- let outTok = 0;
772
- let cost = 0;
773
- let rolloutsWithoutCost = 0;
774
- const rewards = [];
775
- for (const line of lines) {
776
- splits[line.task.split] += 1;
777
- inTok += line.cost.tokens_in ?? 0;
778
- outTok += line.cost.tokens_out ?? 0;
779
- if (line.cost.usd === null) rolloutsWithoutCost++;
780
- else cost += line.cost.usd;
781
- const rw = trainableLineReward(line);
782
- if (rw !== null) rewards.push(rw);
783
- }
784
- return {
785
- records: lines.length,
786
- scoredRecords: rewards.length,
787
- splits,
788
- reward: computeRewardStats(rewards),
789
- models: distinct(lines.map((l) => l.policy.model)),
790
- promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),
791
- commitShas: distinct(lines.map((l) => l.policy.profile_commit)),
792
- totalTokens: { input: inTok, output: outTok },
793
- totalCostUsd: cost,
794
- rolloutsWithoutCost
795
- };
796
- }
1002
+ const splits = {
1003
+ search: 0,
1004
+ dev: 0,
1005
+ holdout: 0,
1006
+ canary: 0
1007
+ };
1008
+ let inTok = 0;
1009
+ let outTok = 0;
1010
+ let cost = 0;
1011
+ let rolloutsWithoutCost = 0;
1012
+ const rewards = [];
1013
+ for (const line of lines) {
1014
+ splits[line.task.split] += 1;
1015
+ inTok += line.cost.tokens_in ?? 0;
1016
+ outTok += line.cost.tokens_out ?? 0;
1017
+ if (line.cost.usd === null) rolloutsWithoutCost++;
1018
+ else cost += line.cost.usd;
1019
+ const rw = trainableLineReward(line);
1020
+ if (rw !== null) rewards.push(rw);
1021
+ }
1022
+ return {
1023
+ records: lines.length,
1024
+ scoredRecords: rewards.length,
1025
+ splits,
1026
+ reward: computeRewardStats(rewards),
1027
+ models: distinct(lines.map((l) => l.policy.model)),
1028
+ promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),
1029
+ commitShas: distinct(lines.map((l) => l.policy.profile_commit)),
1030
+ totalTokens: {
1031
+ input: inTok,
1032
+ output: outTok
1033
+ },
1034
+ totalCostUsd: cost,
1035
+ rolloutsWithoutCost
1036
+ };
1037
+ }
1038
+ /**
1039
+ * Package graded rollout lines into a publishable RL dataset bundle: the
1040
+ * trainer-format JSONL files + a manifest + a datasheet. DPO requires
1041
+ * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
1042
+ * the lines directly via the supplied lookups. Throws on an empty corpus —
1043
+ * an empty dataset must never be published.
1044
+ */
797
1045
  async function buildRlDataset(lines, lookups, config, preferences) {
798
- if (lines.length === 0) {
799
- throw new Error("buildRlDataset: no rollout lines \u2014 refusing to package an empty dataset");
800
- }
801
- const formats = validateDatasetFormats(config.formats === void 0 ? ["sft"] : config.formats);
802
- const files = {};
803
- const rowCounts = {};
804
- if (formats.includes("grpo")) {
805
- const rows = await toGrpoRows(lines, lookups);
806
- requireRows("grpo", rows.length);
807
- files["train.grpo.jsonl"] = toGrpoJsonl(rows);
808
- rowCounts.grpo = rows.length;
809
- }
810
- if (formats.includes("sft")) {
811
- const rows = await toSftRows(lines, lookups);
812
- requireRows("sft", rows.length);
813
- files["train.sft.jsonl"] = toSftJsonl(rows);
814
- rowCounts.sft = rows.length;
815
- }
816
- if (formats.includes("dpo")) {
817
- if (!preferences) {
818
- throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
819
- }
820
- const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines });
821
- requireRows("dpo", rows.length);
822
- files["train.dpo.jsonl"] = toDpoJsonl(rows);
823
- rowCounts.dpo = rows.length;
824
- }
825
- if (!Object.keys(files).some((name) => name.startsWith("train.") && name.endsWith(".jsonl"))) {
826
- throw new Error("buildRlDataset: no trainer file was emitted");
827
- }
828
- const manifest = {
829
- ...config,
830
- formats,
831
- rowCounts,
832
- stats: computeStatsFromLines(lines)
833
- };
834
- files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}
835
- `;
836
- files["DATASHEET.md"] = datasheetToMarkdown(manifest);
837
- return { manifest, files };
1046
+ if (lines.length === 0) throw new Error("buildRlDataset: no rollout lines — refusing to package an empty dataset");
1047
+ const formats = validateDatasetFormats(config.formats === void 0 ? ["sft"] : config.formats);
1048
+ const files = {};
1049
+ const rowCounts = {};
1050
+ if (formats.includes("grpo")) {
1051
+ const rows = await toGrpoRows(lines, lookups);
1052
+ requireRows("grpo", rows.length);
1053
+ files["train.grpo.jsonl"] = toGrpoJsonl(rows);
1054
+ rowCounts.grpo = rows.length;
1055
+ }
1056
+ if (formats.includes("sft")) {
1057
+ const rows = await toSftRows(lines, lookups);
1058
+ requireRows("sft", rows.length);
1059
+ files["train.sft.jsonl"] = toSftJsonl(rows);
1060
+ rowCounts.sft = rows.length;
1061
+ }
1062
+ if (formats.includes("dpo")) {
1063
+ if (!preferences) throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
1064
+ const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines });
1065
+ requireRows("dpo", rows.length);
1066
+ files["train.dpo.jsonl"] = toDpoJsonl(rows);
1067
+ rowCounts.dpo = rows.length;
1068
+ }
1069
+ if (!Object.keys(files).some((name) => name.startsWith("train.") && name.endsWith(".jsonl"))) throw new Error("buildRlDataset: no trainer file was emitted");
1070
+ const manifest = {
1071
+ ...config,
1072
+ formats,
1073
+ rowCounts,
1074
+ stats: computeStatsFromLines(lines)
1075
+ };
1076
+ files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}\n`;
1077
+ files["DATASHEET.md"] = datasheetToMarkdown(manifest);
1078
+ return {
1079
+ manifest,
1080
+ files
1081
+ };
838
1082
  }
839
1083
  function requireRows(format, rows) {
840
- if (rows === 0) {
841
- throw new Error(`buildRlDataset: requested '${format}' format produced no trainable rows`);
842
- }
1084
+ if (rows === 0) throw new Error(`buildRlDataset: requested '${format}' format produced no trainable rows`);
843
1085
  }
844
1086
  function pct(x) {
845
- return `${(x * 100).toFixed(1)}%`;
1087
+ return `${(x * 100).toFixed(1)}%`;
846
1088
  }
847
1089
  function stat(value) {
848
- return value === null ? "n/a" : value.toFixed(3);
1090
+ return value === null ? "n/a" : value.toFixed(3);
849
1091
  }
1092
+ /** Render the "Datasheet for Datasets" card that a buyer reads. */
850
1093
  function datasheetToMarkdown(m) {
851
- const s = m.stats;
852
- const total = s.records || 1;
853
- const splitLines = ["search", "dev", "holdout", "canary"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
854
- const costNote = s.rolloutsWithoutCost > 0 ? ` (floor \u2014 ${s.rolloutsWithoutCost} rollout(s) never captured a cost)` : "";
855
- const deterministic = m.reward.kind === "deterministic";
856
- return [
857
- `# Dataset: ${m.name} \`v${m.version}\``,
858
- "",
859
- `**Domain:** ${m.domain} | **Created:** ${m.createdAtIso} | **License:** ${m.license}`,
860
- "",
861
- "## Reward provenance",
862
- `- **Kind:** ${m.reward.kind}${deterministic ? " (decidable, not judge noise)" : ""}`,
863
- `- **Source:** ${m.reward.source}`,
864
- `- **Description:** ${m.reward.description}`,
865
- "",
866
- "## Composition",
867
- `- **Records (trajectories):** ${s.records}`,
868
- `- **Scored records:** ${s.scoredRecords}`,
869
- `- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(", ")}`,
870
- "- **Splits:**",
871
- splitLines,
872
- "",
873
- "## Reward distribution",
874
- `- n=${s.reward.n} | mean=${stat(s.reward.mean)} | median=${stat(s.reward.median)} | min=${stat(s.reward.min)} | max=${stat(s.reward.max)} | std=${stat(s.reward.std)}`,
875
- "",
876
- "## Provenance",
877
- `- **Models:** ${s.models.join(", ")}`,
878
- `- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
879
- `- **Commits:** ${s.commitShas.join(", ")}`,
880
- `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,
881
- "",
882
- "## Quality gates",
883
- `- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
884
- `- Dedup: ${m.qualityGates?.dedup ? "yes" : "no"} | Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? "yes" : "no"}`,
885
- "",
886
- "## Recommended uses",
887
- m.intendedUse,
888
- "",
889
- "## Out of scope",
890
- m.outOfScope,
891
- "",
892
- "## Limitations",
893
- m.limitations,
894
- "",
895
- "## Token rendering",
896
- "For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns. See `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.",
897
- ""
898
- ].join("\n");
899
- }
900
-
901
- // src/rl/corpus.ts
1094
+ const s = m.stats;
1095
+ const total = s.records || 1;
1096
+ const splitLines = [
1097
+ "search",
1098
+ "dev",
1099
+ "holdout",
1100
+ "canary"
1101
+ ].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
1102
+ const costNote = s.rolloutsWithoutCost > 0 ? ` (floor — ${s.rolloutsWithoutCost} rollout(s) never captured a cost)` : "";
1103
+ const deterministic = m.reward.kind === "deterministic";
1104
+ return [
1105
+ `# Dataset: ${m.name} \`v${m.version}\``,
1106
+ "",
1107
+ `**Domain:** ${m.domain} | **Created:** ${m.createdAtIso} | **License:** ${m.license}`,
1108
+ "",
1109
+ "## Reward provenance",
1110
+ `- **Kind:** ${m.reward.kind}${deterministic ? " (decidable, not judge noise)" : ""}`,
1111
+ `- **Source:** ${m.reward.source}`,
1112
+ `- **Description:** ${m.reward.description}`,
1113
+ "",
1114
+ "## Composition",
1115
+ `- **Records (trajectories):** ${s.records}`,
1116
+ `- **Scored records:** ${s.scoredRecords}`,
1117
+ `- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(", ")}`,
1118
+ "- **Splits:**",
1119
+ splitLines,
1120
+ "",
1121
+ "## Reward distribution",
1122
+ `- n=${s.reward.n} | mean=${stat(s.reward.mean)} | median=${stat(s.reward.median)} | min=${stat(s.reward.min)} | max=${stat(s.reward.max)} | std=${stat(s.reward.std)}`,
1123
+ "",
1124
+ "## Provenance",
1125
+ `- **Models:** ${s.models.join(", ")}`,
1126
+ `- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
1127
+ `- **Commits:** ${s.commitShas.join(", ")}`,
1128
+ `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,
1129
+ "",
1130
+ "## Quality gates",
1131
+ `- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
1132
+ `- Dedup: ${m.qualityGates?.dedup ? "yes" : "no"} | Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? "yes" : "no"}`,
1133
+ "",
1134
+ "## Recommended uses",
1135
+ m.intendedUse,
1136
+ "",
1137
+ "## Out of scope",
1138
+ m.outOfScope,
1139
+ "",
1140
+ "## Limitations",
1141
+ m.limitations,
1142
+ "",
1143
+ "## Token rendering",
1144
+ "For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns. See `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.",
1145
+ ""
1146
+ ].join("\n");
1147
+ }
1148
+ //#endregion
1149
+ //#region src/rl/corpus.ts
1150
+ /**
1151
+ * RL corpus — the durable, append-only accumulation of graded RunRecords that
1152
+ * every eval run deposits BY DEFAULT.
1153
+ *
1154
+ * The dataset is the free exhaust of the normal eval process: we run evals
1155
+ * constantly to get an agent production-ready, and those runs already produce
1156
+ * graded trajectories. Instead of writing them to an ephemeral run dir and
1157
+ * throwing them away, `appendToCorpus` accumulates them into a durable corpus;
1158
+ * `buildDatasetFromCorpus` later harvests the whole corpus into a publishable
1159
+ * bundle. No separate data-collection campaign — the data accrues from work we
1160
+ * do anyway. This is the "best things for free by our process" layer.
1161
+ *
1162
+ * Trajectory text rides on the record as top-level `prompt` / `completion`
1163
+ * (what the eval harnesses capture; the RunRecord validator ignores the extra
1164
+ * keys). The harvest reads them directly — no trace store round-trip needed.
1165
+ */
1166
+ /**
1167
+ * Append graded records to the corpus (append-only JSONL). Deduplicates by
1168
+ * `runId` against what's already on disk so re-running the same harness is
1169
+ * idempotent. Creates the file and parent dir. This is the call every eval
1170
+ * harness makes by default after producing its records.
1171
+ */
902
1172
  function appendToCorpus(records, corpusPath) {
903
- mkdirSync(dirname(corpusPath), { recursive: true });
904
- const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : [];
905
- const seen = new Set(existing.map((r) => r.runId));
906
- const lines = [];
907
- let appended = 0;
908
- let skipped = 0;
909
- for (const r of records) {
910
- if (seen.has(r.runId)) {
911
- skipped++;
912
- continue;
913
- }
914
- seen.add(r.runId);
915
- lines.push(JSON.stringify(r));
916
- appended++;
917
- }
918
- if (lines.length > 0) appendFileSync(corpusPath, `${lines.join("\n")}
919
- `);
920
- return { appended, skipped, total: existing.length + appended };
921
- }
1173
+ mkdirSync(dirname(corpusPath), { recursive: true });
1174
+ const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : [];
1175
+ const seen = new Set(existing.map((r) => r.runId));
1176
+ const lines = [];
1177
+ let appended = 0;
1178
+ let skipped = 0;
1179
+ for (const r of records) {
1180
+ if (seen.has(r.runId)) {
1181
+ skipped++;
1182
+ continue;
1183
+ }
1184
+ seen.add(r.runId);
1185
+ lines.push(JSON.stringify(r));
1186
+ appended++;
1187
+ }
1188
+ if (lines.length > 0) appendFileSync(corpusPath, `${lines.join("\n")}\n`);
1189
+ return {
1190
+ appended,
1191
+ skipped,
1192
+ total: existing.length + appended
1193
+ };
1194
+ }
1195
+ /** Read the full corpus. Returns [] if the corpus does not exist yet. */
922
1196
  function readCorpus(corpusPath) {
923
- if (!existsSync(corpusPath)) return [];
924
- const out = [];
925
- for (const line of readFileSync(corpusPath, "utf8").split("\n")) {
926
- if (line.trim()) out.push(JSON.parse(line));
927
- }
928
- return out;
929
- }
1197
+ if (!existsSync(corpusPath)) return [];
1198
+ const out = [];
1199
+ for (const line of readFileSync(corpusPath, "utf8").split("\n")) if (line.trim()) out.push(JSON.parse(line));
1200
+ return out;
1201
+ }
1202
+ /**
1203
+ * The harvest's score reader is GATED: a gamed run reads 0, so it cannot buy
1204
+ * its way past `minScore` into the published bundle with its claimed score.
1205
+ * `null` = unscored (a labeled gap, dropped before packaging, never a 0).
1206
+ */
930
1207
  function rewardOf(r) {
931
- const v = trainingScore(r);
932
- return typeof v === "number" && Number.isFinite(v) ? v : null;
933
- }
1208
+ const v = trainingScore(r);
1209
+ return typeof v === "number" && Number.isFinite(v) ? v : null;
1210
+ }
1211
+ /**
1212
+ * Harvest the accumulated corpus into a publishable RL dataset bundle. Reads
1213
+ * trajectory text from each record's top-level `prompt`/`completion`; records
1214
+ * missing either are excluded (a graded score with no trajectory can't train).
1215
+ * Optionally filters by score / split. Throws (via buildRlDataset) if nothing
1216
+ * survives — an empty dataset must never be published.
1217
+ *
1218
+ * `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run
1219
+ * cannot buy its way into the published bundle with its claimed score —
1220
+ * `minScore` is exactly the door a reward-hacked run would otherwise clear for
1221
+ * SFT. Unscored records are dropped before packaging: a missing label is not a
1222
+ * zero, and it is not publishable either.
1223
+ */
934
1224
  async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
935
- let records = readCorpus(corpusPath).filter(
936
- (r) => typeof r.prompt === "string" && typeof r.completion === "string"
937
- );
938
- if (opts.splits) records = records.filter((r) => opts.splits.includes(r.splitTag));
939
- records = records.filter((r) => rewardOf(r) !== null);
940
- if (opts.minScore != null) {
941
- records = records.filter((r) => {
942
- const reward = rewardOf(r);
943
- return reward !== null && reward >= opts.minScore;
944
- });
945
- }
946
- const text = new Map(
947
- records.map((r) => [r.runId, { prompt: r.prompt, completion: r.completion }])
948
- );
949
- const lookups = {
950
- promptOf: (id) => text.get(id)?.prompt ?? "",
951
- completionOf: (id) => text.get(id)?.completion ?? "",
952
- allowHeldOutTrainingData: opts.allowHeldOutTrainingData
953
- };
954
- const { rows } = await mintRolloutRows(records, new InMemoryTraceStore());
955
- return buildRlDataset(rows, lookups, config);
956
- }
957
-
958
- // src/rl/predictive-validity-researcher.ts
1225
+ let records = readCorpus(corpusPath).filter((r) => typeof r.prompt === "string" && typeof r.completion === "string");
1226
+ if (opts.splits) records = records.filter((r) => opts.splits.includes(r.splitTag));
1227
+ records = records.filter((r) => rewardOf(r) !== null);
1228
+ if (opts.minScore != null) records = records.filter((r) => {
1229
+ const reward = rewardOf(r);
1230
+ return reward !== null && reward >= opts.minScore;
1231
+ });
1232
+ const text = new Map(records.map((r) => [r.runId, {
1233
+ prompt: r.prompt,
1234
+ completion: r.completion
1235
+ }]));
1236
+ const lookups = {
1237
+ promptOf: (id) => text.get(id)?.prompt ?? "",
1238
+ completionOf: (id) => text.get(id)?.completion ?? "",
1239
+ allowHeldOutTrainingData: opts.allowHeldOutTrainingData
1240
+ };
1241
+ const { rows } = await mintRolloutRows(records, new InMemoryTraceStore());
1242
+ return buildRlDataset(rows, lookups, config);
1243
+ }
1244
+ //#endregion
1245
+ //#region src/rl/predictive-validity-researcher.ts
1246
+ /**
1247
+ * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
1248
+ * rubrics that don't predict deployment outcomes don't earn weight.
1249
+ */
959
1250
  var PredictiveValidityResearcher = class {
960
- opts;
961
- lastReport = null;
962
- constructor(opts) {
963
- this.opts = opts;
964
- }
965
- async inspectFailures(runs) {
966
- const threshold = this.opts.failureThreshold ?? 0.5;
967
- const failures = [];
968
- const failingRuns = runs.filter((r) => {
969
- const score = runTaskScore(r);
970
- return typeof score === "number" && score < threshold;
971
- });
972
- if (failingRuns.length === 0) return failures;
973
- const grouped = /* @__PURE__ */ new Map();
974
- for (const r of failingRuns) {
975
- const arr = grouped.get(r.candidateId) ?? [];
976
- arr.push(r);
977
- grouped.set(r.candidateId, arr);
978
- }
979
- for (const [candidateId, group] of grouped.entries()) {
980
- const meanScore = group.reduce((s, r) => {
981
- const score = runTaskScore(r);
982
- if (score === void 0) {
983
- throw new Error(`failing run ${r.runId} unexpectedly has no task score`);
984
- }
985
- return s + score;
986
- }, 0) / group.length;
987
- failures.push({
988
- code: `low-score-${candidateId}`,
989
- description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,
990
- evidence: {
991
- runIds: group.slice(0, 8).map((r) => r.runId),
992
- samples: group.length
993
- }
994
- });
995
- }
996
- return failures;
997
- }
998
- async proposeChange(failures) {
999
- if (failures.length === 0) return [];
1000
- if (this.lastReport === null) {
1001
- return [
1002
- {
1003
- kind: "threshold",
1004
- payload: { directive: "researcher.collect-more-outcomes" },
1005
- rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
1006
- }
1007
- ];
1008
- }
1009
- const decorativeThreshold = this.opts.decorativeThreshold ?? 0.4;
1010
- const changes = [];
1011
- for (const ranking of this.lastReport.ranked) {
1012
- if (ranking.verdict === "load_bearing") continue;
1013
- if (Math.abs(ranking.spearman) >= decorativeThreshold) continue;
1014
- changes.push({
1015
- kind: "reviewer_prompt",
1016
- payload: {
1017
- rubric: ranking.rubric,
1018
- action: "down-weight",
1019
- spearman: ranking.spearman,
1020
- bestOutcome: ranking.bestOutcome
1021
- },
1022
- rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,
1023
- expectedDelta: -Math.max(0, 0.05 - Math.abs(ranking.spearman))
1024
- });
1025
- }
1026
- for (const ranking of this.lastReport.ranked.slice(0, 1)) {
1027
- if (ranking.verdict !== "load_bearing") continue;
1028
- changes.push({
1029
- kind: "reviewer_prompt",
1030
- payload: {
1031
- rubric: ranking.rubric,
1032
- action: "up-weight",
1033
- spearman: ranking.spearman,
1034
- bestOutcome: ranking.bestOutcome
1035
- },
1036
- rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,
1037
- expectedDelta: Math.max(0, Math.abs(ranking.spearman) - 0.5) * 0.1
1038
- });
1039
- }
1040
- return changes;
1041
- }
1042
- async applyChange(changes, baseline) {
1043
- return {
1044
- ...baseline,
1045
- changes: [...baseline.changes, ...changes]
1046
- };
1047
- }
1048
- async evaluateChange(plan) {
1049
- const emptyGate = {
1050
- promote: false,
1051
- candidateId: plan.proposedCandidateId,
1052
- baselineId: plan.baselineCandidateId,
1053
- evidence: {
1054
- productiveRuns: 0,
1055
- unpairedCandidateRuns: 0,
1056
- unpairedBaselineRuns: 0,
1057
- medianPairedDelta: null,
1058
- pairedCI: null,
1059
- pairedPValue: null,
1060
- searchScore: null,
1061
- holdoutScore: null,
1062
- overfitGap: null,
1063
- baselineOverfitGap: null,
1064
- medianCandidateCost: null,
1065
- medianBaselineCost: null,
1066
- realnessGatedRuns: 0
1067
- },
1068
- reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
1069
- rejectionCode: "few_runs"
1070
- };
1071
- return {
1072
- plan,
1073
- runs: [],
1074
- gateDecision: emptyGate
1075
- };
1076
- }
1077
- /**
1078
- * Run the predictive-validity check explicitly against a fresh RunRecord
1079
- * set. Updates the researcher's cached report so subsequent
1080
- * `proposeChange` calls have evidence to draw from.
1081
- */
1082
- async runValidityCheck(runs) {
1083
- const report = await rubricPredictiveValidity({
1084
- runs,
1085
- outcomes: this.opts.outcomes,
1086
- outcomeMetrics: this.opts.outcomeMetrics,
1087
- rubrics: this.opts.rubrics
1088
- });
1089
- if (this.opts.onReport) await this.opts.onReport(report);
1090
- this.lastReport = report;
1091
- return report;
1092
- }
1093
- /**
1094
- * Force-feed a predictive-validity report into the researcher state —
1095
- * useful when the consumer ran the report out-of-band and wants the
1096
- * researcher's later proposals informed by it.
1097
- */
1098
- setReport(report) {
1099
- this.lastReport = report;
1100
- }
1101
- getLastReport() {
1102
- return this.lastReport;
1103
- }
1251
+ opts;
1252
+ lastReport = null;
1253
+ constructor(opts) {
1254
+ this.opts = opts;
1255
+ }
1256
+ async inspectFailures(runs) {
1257
+ const threshold = this.opts.failureThreshold ?? .5;
1258
+ const failures = [];
1259
+ const failingRuns = runs.filter((r) => {
1260
+ const score = runTaskScore(r);
1261
+ return typeof score === "number" && score < threshold;
1262
+ });
1263
+ if (failingRuns.length === 0) return failures;
1264
+ const grouped = /* @__PURE__ */ new Map();
1265
+ for (const r of failingRuns) {
1266
+ const arr = grouped.get(r.candidateId) ?? [];
1267
+ arr.push(r);
1268
+ grouped.set(r.candidateId, arr);
1269
+ }
1270
+ for (const [candidateId, group] of grouped.entries()) {
1271
+ const meanScore = group.reduce((s, r) => {
1272
+ const score = runTaskScore(r);
1273
+ if (score === void 0) throw new Error(`failing run ${r.runId} unexpectedly has no task score`);
1274
+ return s + score;
1275
+ }, 0) / group.length;
1276
+ failures.push({
1277
+ code: `low-score-${candidateId}`,
1278
+ description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,
1279
+ evidence: {
1280
+ runIds: group.slice(0, 8).map((r) => r.runId),
1281
+ samples: group.length
1282
+ }
1283
+ });
1284
+ }
1285
+ return failures;
1286
+ }
1287
+ async proposeChange(failures) {
1288
+ if (failures.length === 0) return [];
1289
+ if (this.lastReport === null) return [{
1290
+ kind: "threshold",
1291
+ payload: { directive: "researcher.collect-more-outcomes" },
1292
+ rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
1293
+ }];
1294
+ const decorativeThreshold = this.opts.decorativeThreshold ?? .4;
1295
+ const changes = [];
1296
+ for (const ranking of this.lastReport.ranked) {
1297
+ if (ranking.verdict === "load_bearing") continue;
1298
+ if (Math.abs(ranking.spearman) >= decorativeThreshold) continue;
1299
+ changes.push({
1300
+ kind: "reviewer_prompt",
1301
+ payload: {
1302
+ rubric: ranking.rubric,
1303
+ action: "down-weight",
1304
+ spearman: ranking.spearman,
1305
+ bestOutcome: ranking.bestOutcome
1306
+ },
1307
+ rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,
1308
+ expectedDelta: -Math.max(0, .05 - Math.abs(ranking.spearman))
1309
+ });
1310
+ }
1311
+ for (const ranking of this.lastReport.ranked.slice(0, 1)) {
1312
+ if (ranking.verdict !== "load_bearing") continue;
1313
+ changes.push({
1314
+ kind: "reviewer_prompt",
1315
+ payload: {
1316
+ rubric: ranking.rubric,
1317
+ action: "up-weight",
1318
+ spearman: ranking.spearman,
1319
+ bestOutcome: ranking.bestOutcome
1320
+ },
1321
+ rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,
1322
+ expectedDelta: Math.max(0, Math.abs(ranking.spearman) - .5) * .1
1323
+ });
1324
+ }
1325
+ return changes;
1326
+ }
1327
+ async applyChange(changes, baseline) {
1328
+ return {
1329
+ ...baseline,
1330
+ changes: [...baseline.changes, ...changes]
1331
+ };
1332
+ }
1333
+ async evaluateChange(plan) {
1334
+ return {
1335
+ plan,
1336
+ runs: [],
1337
+ gateDecision: {
1338
+ promote: false,
1339
+ candidateId: plan.proposedCandidateId,
1340
+ baselineId: plan.baselineCandidateId,
1341
+ evidence: {
1342
+ productiveRuns: 0,
1343
+ unpairedCandidateRuns: 0,
1344
+ unpairedBaselineRuns: 0,
1345
+ medianPairedDelta: null,
1346
+ pairedCI: null,
1347
+ pairedPValue: null,
1348
+ searchScore: null,
1349
+ holdoutScore: null,
1350
+ overfitGap: null,
1351
+ baselineOverfitGap: null,
1352
+ medianCandidateCost: null,
1353
+ medianBaselineCost: null,
1354
+ realnessGatedRuns: 0
1355
+ },
1356
+ reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
1357
+ rejectionCode: "few_runs"
1358
+ }
1359
+ };
1360
+ }
1361
+ /**
1362
+ * Run the predictive-validity check explicitly against a fresh RunRecord
1363
+ * set. Updates the researcher's cached report so subsequent
1364
+ * `proposeChange` calls have evidence to draw from.
1365
+ */
1366
+ async runValidityCheck(runs) {
1367
+ const report = await rubricPredictiveValidity({
1368
+ runs,
1369
+ outcomes: this.opts.outcomes,
1370
+ outcomeMetrics: this.opts.outcomeMetrics,
1371
+ rubrics: this.opts.rubrics
1372
+ });
1373
+ if (this.opts.onReport) await this.opts.onReport(report);
1374
+ this.lastReport = report;
1375
+ return report;
1376
+ }
1377
+ /**
1378
+ * Force-feed a predictive-validity report into the researcher state —
1379
+ * useful when the consumer ran the report out-of-band and wants the
1380
+ * researcher's later proposals informed by it.
1381
+ */
1382
+ setReport(report) {
1383
+ this.lastReport = report;
1384
+ }
1385
+ getLastReport() {
1386
+ return this.lastReport;
1387
+ }
1104
1388
  };
1105
-
1106
- // src/rl/preferences.ts
1107
- var SPLIT_DEFAULT = "search";
1389
+ //#endregion
1390
+ //#region src/rl/preferences.ts
1391
+ /** The split each path pairs by default: training data comes from search. */
1392
+ const SPLIT_DEFAULT = "search";
1393
+ /**
1394
+ * Convert rollout lines to preference triples for RL training.
1395
+ *
1396
+ * Returns a structured report so callers can see how much data was
1397
+ * dropped and why (low-margin pairs, singleton cells). For production
1398
+ * pipelines, you usually want to:
1399
+ *
1400
+ * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
1401
+ * 2. Mint the runs with `mintRolloutRows` and call this with
1402
+ * `strategy: 'paired-by-scenario-and-seed'`
1403
+ * 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with
1404
+ * prompt/completion resolvers and pipe to your DPO trainer
1405
+ *
1406
+ * The gate is what makes a preference dataset safe: ordered on an ungated
1407
+ * score, a gamed run with an inflated number becomes the `chosen` side and DPO
1408
+ * is trained to prefer the gaming trajectory over its honest sibling. A gated
1409
+ * line arrives here already scored 0, so it sinks to `rejected`.
1410
+ */
1108
1411
  function extractPreferences(lines, opts = {}) {
1109
- const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
1110
- const minMargin = opts.minMargin ?? 0.05;
1111
- const requestedSplit = opts.split;
1112
- if (requestedSplit === "holdout" && opts.allowHeldOutTrainingData !== true) {
1113
- throw new Error('extractPreferences: split "holdout" requires allowHeldOutTrainingData: true');
1114
- }
1115
- if (requestedSplit === "dev" || requestedSplit === "canary") {
1116
- throw new Error(
1117
- `extractPreferences: split "${requestedSplit}" is evaluation-only; train from "search"`
1118
- );
1119
- }
1120
- const candidates = candidatesFromLines(lines, opts);
1121
- const report = pairCandidates(candidates.rows, strategy, minMargin);
1122
- return { ...report, linesWithoutCandidateId: candidates.withoutCandidateId };
1412
+ const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
1413
+ const minMargin = opts.minMargin ?? .05;
1414
+ const requestedSplit = opts.split;
1415
+ if (requestedSplit === "holdout" && opts.allowHeldOutTrainingData !== true) throw new Error("extractPreferences: split \"holdout\" requires allowHeldOutTrainingData: true");
1416
+ if (requestedSplit === "dev" || requestedSplit === "canary") throw new Error(`extractPreferences: split "${requestedSplit}" is evaluation-only; train from "search"`);
1417
+ const candidates = candidatesFromLines(lines, opts);
1418
+ return {
1419
+ ...pairCandidates(candidates.rows, strategy, minMargin),
1420
+ linesWithoutCandidateId: candidates.withoutCandidateId
1421
+ };
1123
1422
  }
1124
1423
  function candidatesFromLines(lines, opts) {
1125
- const split = opts.split ?? SPLIT_DEFAULT;
1126
- const rows = [];
1127
- let withoutCandidateId = 0;
1128
- for (const line of lines) {
1129
- if (line.task.split !== split) continue;
1130
- if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {
1131
- continue;
1132
- }
1133
- const score = trainableLineReward(line);
1134
- if (score === null) continue;
1135
- const candidateId = line.candidate_id;
1136
- if (candidateId === null || candidateId === void 0 || candidateId.length === 0) {
1137
- withoutCandidateId++;
1138
- continue;
1139
- }
1140
- rows.push({
1141
- scenarioId: line.task.instance_id,
1142
- runId: line.run_id,
1143
- candidateId,
1144
- seed: line.task.seed,
1145
- score,
1146
- // `policy.*` is nullable on the wire; a minted line always carries these
1147
- // (RunRecord makes them mandatory). Empty string marks "not recorded" so
1148
- // `toTRLFormat`'s hash lookup fails visibly instead of silently matching.
1149
- promptHash: line.policy.prompt_hash ?? "",
1150
- configHash: line.policy.config_hash ?? "",
1151
- model: line.policy.model ?? ""
1152
- });
1153
- }
1154
- return { rows, withoutCandidateId };
1424
+ const split = opts.split ?? SPLIT_DEFAULT;
1425
+ const rows = [];
1426
+ let withoutCandidateId = 0;
1427
+ for (const line of lines) {
1428
+ if (line.task.split !== split) continue;
1429
+ if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) continue;
1430
+ const score = trainableLineReward(line);
1431
+ if (score === null) continue;
1432
+ const candidateId = line.candidate_id;
1433
+ if (candidateId === null || candidateId === void 0 || candidateId.length === 0) {
1434
+ withoutCandidateId++;
1435
+ continue;
1436
+ }
1437
+ rows.push({
1438
+ scenarioId: line.task.instance_id,
1439
+ runId: line.run_id,
1440
+ candidateId,
1441
+ seed: line.task.seed,
1442
+ score,
1443
+ promptHash: line.policy.prompt_hash ?? "",
1444
+ configHash: line.policy.config_hash ?? "",
1445
+ model: line.policy.model ?? ""
1446
+ });
1447
+ }
1448
+ return {
1449
+ rows,
1450
+ withoutCandidateId
1451
+ };
1155
1452
  }
1156
1453
  function pairCandidates(scoredEntries, strategy, minMargin) {
1157
- const pairs = [];
1158
- let pairsBelowMargin = 0;
1159
- let cellsSingleton = 0;
1160
- let cellsInspected = 0;
1161
- if (strategy === "paired-by-scenario-and-seed") {
1162
- const groups = /* @__PURE__ */ new Map();
1163
- for (const e of scoredEntries) {
1164
- const key = `${e.scenarioId}::${e.seed}`;
1165
- const arr = groups.get(key) ?? [];
1166
- arr.push(e);
1167
- groups.set(key, arr);
1168
- }
1169
- for (const members of groups.values()) {
1170
- cellsInspected++;
1171
- if (members.length < 2) {
1172
- cellsSingleton++;
1173
- continue;
1174
- }
1175
- for (let i = 0; i < members.length; i++) {
1176
- for (let j = i + 1; j < members.length; j++) {
1177
- const a = members[i];
1178
- const b = members[j];
1179
- if (a.candidateId === b.candidateId) continue;
1180
- const result = makePair(a, b, a.scenarioId, minMargin);
1181
- if (result.kind === "admit") pairs.push(result.pair);
1182
- else pairsBelowMargin++;
1183
- }
1184
- }
1185
- }
1186
- } else if (strategy === "paired-by-scenario") {
1187
- const byScenarioVariant = /* @__PURE__ */ new Map();
1188
- for (const e of scoredEntries) {
1189
- let perScenario = byScenarioVariant.get(e.scenarioId);
1190
- if (!perScenario) {
1191
- perScenario = /* @__PURE__ */ new Map();
1192
- byScenarioVariant.set(e.scenarioId, perScenario);
1193
- }
1194
- const cur = perScenario.get(e.candidateId);
1195
- if (cur) {
1196
- cur.sum += e.score;
1197
- cur.n++;
1198
- } else perScenario.set(e.candidateId, { entry: e, sum: e.score, n: 1 });
1199
- }
1200
- for (const [sid, perVariant] of byScenarioVariant.entries()) {
1201
- cellsInspected++;
1202
- const arr = [...perVariant.values()].map((agg) => ({
1203
- ...agg.entry,
1204
- score: agg.sum / agg.n
1205
- }));
1206
- if (arr.length < 2) {
1207
- cellsSingleton++;
1208
- continue;
1209
- }
1210
- for (let i = 0; i < arr.length; i++) {
1211
- for (let j = i + 1; j < arr.length; j++) {
1212
- const result = makePair(arr[i], arr[j], sid, minMargin);
1213
- if (result.kind === "admit") pairs.push(result.pair);
1214
- else pairsBelowMargin++;
1215
- }
1216
- }
1217
- }
1218
- } else {
1219
- const byScenario = /* @__PURE__ */ new Map();
1220
- for (const e of scoredEntries) {
1221
- const arr = byScenario.get(e.scenarioId) ?? [];
1222
- arr.push(e);
1223
- byScenario.set(e.scenarioId, arr);
1224
- }
1225
- for (const [sid, arr] of byScenario.entries()) {
1226
- cellsInspected++;
1227
- if (arr.length < 2) {
1228
- cellsSingleton++;
1229
- continue;
1230
- }
1231
- const sorted = [...arr].sort((a, b) => a.score - b.score);
1232
- const top = sorted[sorted.length - 1];
1233
- const bot = sorted[0];
1234
- if (top.candidateId === bot.candidateId) {
1235
- cellsSingleton++;
1236
- continue;
1237
- }
1238
- const result = makePair(bot, top, sid, minMargin);
1239
- if (result.kind === "admit") pairs.push(result.pair);
1240
- else pairsBelowMargin++;
1241
- }
1242
- }
1243
- return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
1244
- }
1245
- var PREFERENCE_RUN_IDS = (t) => [
1246
- t.chosenRunId,
1247
- t.rejectedRunId
1248
- ];
1249
- var TRL_CONTEXT_REQUIREMENT = {
1250
- exporter: "TRL preference export",
1251
- contextType: "RolloutLineContext",
1252
- because: "a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair."
1454
+ const pairs = [];
1455
+ let pairsBelowMargin = 0;
1456
+ let cellsSingleton = 0;
1457
+ let cellsInspected = 0;
1458
+ if (strategy === "paired-by-scenario-and-seed") {
1459
+ const groups = /* @__PURE__ */ new Map();
1460
+ for (const e of scoredEntries) {
1461
+ const key = `${e.scenarioId}::${e.seed}`;
1462
+ const arr = groups.get(key) ?? [];
1463
+ arr.push(e);
1464
+ groups.set(key, arr);
1465
+ }
1466
+ for (const members of groups.values()) {
1467
+ cellsInspected++;
1468
+ if (members.length < 2) {
1469
+ cellsSingleton++;
1470
+ continue;
1471
+ }
1472
+ for (let i = 0; i < members.length; i++) for (let j = i + 1; j < members.length; j++) {
1473
+ const a = members[i];
1474
+ const b = members[j];
1475
+ if (a.candidateId === b.candidateId) continue;
1476
+ const result = makePair(a, b, a.scenarioId, minMargin);
1477
+ if (result.kind === "admit") pairs.push(result.pair);
1478
+ else pairsBelowMargin++;
1479
+ }
1480
+ }
1481
+ } else if (strategy === "paired-by-scenario") {
1482
+ const byScenarioVariant = /* @__PURE__ */ new Map();
1483
+ for (const e of scoredEntries) {
1484
+ let perScenario = byScenarioVariant.get(e.scenarioId);
1485
+ if (!perScenario) {
1486
+ perScenario = /* @__PURE__ */ new Map();
1487
+ byScenarioVariant.set(e.scenarioId, perScenario);
1488
+ }
1489
+ const cur = perScenario.get(e.candidateId);
1490
+ if (cur) {
1491
+ cur.sum += e.score;
1492
+ cur.n++;
1493
+ } else perScenario.set(e.candidateId, {
1494
+ entry: e,
1495
+ sum: e.score,
1496
+ n: 1
1497
+ });
1498
+ }
1499
+ for (const [sid, perVariant] of byScenarioVariant.entries()) {
1500
+ cellsInspected++;
1501
+ const arr = [...perVariant.values()].map((agg) => ({
1502
+ ...agg.entry,
1503
+ score: agg.sum / agg.n
1504
+ }));
1505
+ if (arr.length < 2) {
1506
+ cellsSingleton++;
1507
+ continue;
1508
+ }
1509
+ for (let i = 0; i < arr.length; i++) for (let j = i + 1; j < arr.length; j++) {
1510
+ const result = makePair(arr[i], arr[j], sid, minMargin);
1511
+ if (result.kind === "admit") pairs.push(result.pair);
1512
+ else pairsBelowMargin++;
1513
+ }
1514
+ }
1515
+ } else {
1516
+ const byScenario = /* @__PURE__ */ new Map();
1517
+ for (const e of scoredEntries) {
1518
+ const arr = byScenario.get(e.scenarioId) ?? [];
1519
+ arr.push(e);
1520
+ byScenario.set(e.scenarioId, arr);
1521
+ }
1522
+ for (const [sid, arr] of byScenario.entries()) {
1523
+ cellsInspected++;
1524
+ if (arr.length < 2) {
1525
+ cellsSingleton++;
1526
+ continue;
1527
+ }
1528
+ const sorted = [...arr].sort((a, b) => a.score - b.score);
1529
+ const top = sorted[sorted.length - 1];
1530
+ const bot = sorted[0];
1531
+ if (top.candidateId === bot.candidateId) {
1532
+ cellsSingleton++;
1533
+ continue;
1534
+ }
1535
+ const result = makePair(bot, top, sid, minMargin);
1536
+ if (result.kind === "admit") pairs.push(result.pair);
1537
+ else pairsBelowMargin++;
1538
+ }
1539
+ }
1540
+ return {
1541
+ pairs,
1542
+ cellsInspected,
1543
+ pairsBelowMargin,
1544
+ cellsSingleton,
1545
+ strategy
1546
+ };
1547
+ }
1548
+ const PREFERENCE_RUN_IDS = (t) => [t.chosenRunId, t.rejectedRunId];
1549
+ const TRL_CONTEXT_REQUIREMENT = {
1550
+ exporter: "TRL preference export",
1551
+ contextType: "RolloutLineContext",
1552
+ because: "a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair."
1253
1553
  };
1254
- var ANTHROPIC_CONTEXT_REQUIREMENT = {
1255
- exporter: "Anthropic preference export",
1256
- contextType: "RolloutLineContext",
1257
- because: "a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one."
1554
+ const ANTHROPIC_CONTEXT_REQUIREMENT = {
1555
+ exporter: "Anthropic preference export",
1556
+ contextType: "RolloutLineContext",
1557
+ because: "a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one."
1258
1558
  };
1559
+ /**
1560
+ * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
1561
+ * where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes
1562
+ * would optimize the policy toward emitting hex digests. Neither the prompt
1563
+ * nor the completions live on the triple (it carries only run ids and hashes),
1564
+ * so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`
1565
+ * takes, keyed by run id, and this function resolves real text.
1566
+ *
1567
+ * The chosen and rejected sides of a valid pair share one prompt; resolving
1568
+ * both and comparing catches lookup bugs (a stale map keyed by the wrong id)
1569
+ * before they ship a row whose prompt does not match its rejected completion.
1570
+ *
1571
+ * `context` is REQUIRED: this is the third exporter over the identical
1572
+ * line-less input class, and the round that hardened `toPrmRows` while leaving
1573
+ * `toDpoRows` open is why every one of them now takes the same argument and
1574
+ * runs the same admission rule.
1575
+ */
1259
1576
  async function toTRLFormat(triples, lookups, context) {
1260
- const admitted = admitUngatedByInvocation(
1261
- triples,
1262
- PREFERENCE_RUN_IDS,
1263
- context,
1264
- TRL_CONTEXT_REQUIREMENT
1265
- );
1266
- const out = [];
1267
- for (const t of admitted) {
1268
- const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
1269
- Promise.resolve(lookups.promptOf(t.chosenRunId)),
1270
- Promise.resolve(lookups.promptOf(t.rejectedRunId)),
1271
- Promise.resolve(lookups.completionOf(t.chosenRunId)),
1272
- Promise.resolve(lookups.completionOf(t.rejectedRunId))
1273
- ]);
1274
- if (chosenPrompt !== rejectedPrompt) {
1275
- throw new Error(
1276
- `toTRLFormat: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`
1277
- );
1278
- }
1279
- out.push({ prompt: chosenPrompt, chosen, rejected });
1280
- }
1281
- return out;
1282
- }
1577
+ const admitted = admitUngatedByInvocation(triples, PREFERENCE_RUN_IDS, context, TRL_CONTEXT_REQUIREMENT);
1578
+ const out = [];
1579
+ for (const t of admitted) {
1580
+ const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
1581
+ Promise.resolve(lookups.promptOf(t.chosenRunId)),
1582
+ Promise.resolve(lookups.promptOf(t.rejectedRunId)),
1583
+ Promise.resolve(lookups.completionOf(t.chosenRunId)),
1584
+ Promise.resolve(lookups.completionOf(t.rejectedRunId))
1585
+ ]);
1586
+ if (chosenPrompt !== rejectedPrompt) throw new Error(`toTRLFormat: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`);
1587
+ out.push({
1588
+ prompt: chosenPrompt,
1589
+ chosen,
1590
+ rejected
1591
+ });
1592
+ }
1593
+ return out;
1594
+ }
1595
+ /**
1596
+ * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
1597
+ * shape. Same caveat as TRL: prompt + outputs are content the caller has
1598
+ * to map back from the run record / raw event log.
1599
+ *
1600
+ * `context` is REQUIRED — see `toTRLFormat`. The emitted `margin` is a number
1601
+ * derived from the two runs' rewards, so this row is training signal even
1602
+ * though it ships no completion text.
1603
+ */
1283
1604
  function toAnthropicFormat(triples, context) {
1284
- return admitUngatedByInvocation(
1285
- triples,
1286
- PREFERENCE_RUN_IDS,
1287
- context,
1288
- ANTHROPIC_CONTEXT_REQUIREMENT
1289
- ).map((t) => ({
1290
- scenarioId: t.scenarioId,
1291
- chosenRunId: t.chosenRunId,
1292
- rejectedRunId: t.rejectedRunId,
1293
- margin: t.marginScore
1294
- }));
1605
+ return admitUngatedByInvocation(triples, PREFERENCE_RUN_IDS, context, ANTHROPIC_CONTEXT_REQUIREMENT).map((t) => ({
1606
+ scenarioId: t.scenarioId,
1607
+ chosenRunId: t.chosenRunId,
1608
+ rejectedRunId: t.rejectedRunId,
1609
+ margin: t.marginScore
1610
+ }));
1295
1611
  }
1296
1612
  function makePair(a, b, scenarioId, minMargin) {
1297
- const margin = Math.abs(a.score - b.score);
1298
- if (margin < minMargin) return { kind: "reject" };
1299
- const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
1300
- const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : void 0;
1301
- return {
1302
- kind: "admit",
1303
- pair: {
1304
- scenarioId,
1305
- chosenRunId: chosen.runId,
1306
- rejectedRunId: rejected.runId,
1307
- chosenVariantId: chosen.candidateId,
1308
- rejectedVariantId: rejected.candidateId,
1309
- marginScore: chosen.score - rejected.score,
1310
- scores: { chosen: chosen.score, rejected: rejected.score },
1311
- seed,
1312
- meta: {
1313
- chosenPromptHash: chosen.promptHash,
1314
- rejectedPromptHash: rejected.promptHash,
1315
- chosenConfigHash: chosen.configHash,
1316
- rejectedConfigHash: rejected.configHash,
1317
- chosenModel: chosen.model,
1318
- rejectedModel: rejected.model
1319
- }
1320
- }
1321
- };
1322
- }
1323
-
1324
- // src/rl/process-reward.ts
1613
+ if (Math.abs(a.score - b.score) < minMargin) return { kind: "reject" };
1614
+ const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
1615
+ const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : void 0;
1616
+ return {
1617
+ kind: "admit",
1618
+ pair: {
1619
+ scenarioId,
1620
+ chosenRunId: chosen.runId,
1621
+ rejectedRunId: rejected.runId,
1622
+ chosenVariantId: chosen.candidateId,
1623
+ rejectedVariantId: rejected.candidateId,
1624
+ marginScore: chosen.score - rejected.score,
1625
+ scores: {
1626
+ chosen: chosen.score,
1627
+ rejected: rejected.score
1628
+ },
1629
+ seed,
1630
+ meta: {
1631
+ chosenPromptHash: chosen.promptHash,
1632
+ rejectedPromptHash: rejected.promptHash,
1633
+ chosenConfigHash: chosen.configHash,
1634
+ rejectedConfigHash: rejected.configHash,
1635
+ chosenModel: chosen.model,
1636
+ rejectedModel: rejected.model
1637
+ }
1638
+ }
1639
+ };
1640
+ }
1641
+ //#endregion
1642
+ //#region src/rl/process-reward.ts
1325
1643
  async function extractStepRewards(store, runId, opts) {
1326
- const spans = await store.spans({ runId });
1327
- const ordered = [...spans].sort((a, b) => a.startedAt - b.startedAt);
1328
- const out = [];
1329
- let idx = 0;
1330
- for (const span of ordered) {
1331
- if (opts.preFilter && !opts.preFilter(span)) continue;
1332
- let scored = null;
1333
- for (const s of opts.scorers) {
1334
- if (!s.appliesTo.includes(span.kind)) continue;
1335
- const r = await s.score(span);
1336
- if (r) {
1337
- scored = r;
1338
- break;
1339
- }
1340
- }
1341
- if (!scored) continue;
1342
- out.push({
1343
- spanId: span.spanId,
1344
- runId,
1345
- stepIndex: idx++,
1346
- kind: span.kind,
1347
- name: span.name,
1348
- reward: scored.reward,
1349
- determinism: scored.determinism,
1350
- rationale: scored.rationale,
1351
- weight: scored.weight
1352
- });
1353
- }
1354
- return out;
1644
+ const ordered = [...await store.spans({ runId })].sort((a, b) => a.startedAt - b.startedAt);
1645
+ const out = [];
1646
+ let idx = 0;
1647
+ for (const span of ordered) {
1648
+ if (opts.preFilter && !opts.preFilter(span)) continue;
1649
+ let scored = null;
1650
+ for (const s of opts.scorers) {
1651
+ if (!s.appliesTo.includes(span.kind)) continue;
1652
+ const r = await s.score(span);
1653
+ if (r) {
1654
+ scored = r;
1655
+ break;
1656
+ }
1657
+ }
1658
+ if (!scored) continue;
1659
+ out.push({
1660
+ spanId: span.spanId,
1661
+ runId,
1662
+ stepIndex: idx++,
1663
+ kind: span.kind,
1664
+ name: span.name,
1665
+ reward: scored.reward,
1666
+ determinism: scored.determinism,
1667
+ rationale: scored.rationale,
1668
+ weight: scored.weight
1669
+ });
1670
+ }
1671
+ return out;
1355
1672
  }
1356
1673
  function runwiseStepRewardSummary(stepRewards) {
1357
- if (stepRewards.length === 0) {
1358
- return {
1359
- runId: "",
1360
- totalSteps: 0,
1361
- meanReward: 0,
1362
- sumWeightedReward: 0,
1363
- failureFraction: 0,
1364
- worstStepDelta: 0,
1365
- worstStepIndex: null
1366
- };
1367
- }
1368
- const runId = stepRewards[0].runId;
1369
- let sumW = 0;
1370
- let sumWR = 0;
1371
- let failures = 0;
1372
- let worstDelta = 0;
1373
- let worstIdx = null;
1374
- let prev = stepRewards[0].reward;
1375
- for (let i = 0; i < stepRewards.length; i++) {
1376
- const s = stepRewards[i];
1377
- const w = s.weight ?? 1;
1378
- sumW += w;
1379
- sumWR += w * s.reward;
1380
- if (s.reward < 0.5) failures++;
1381
- if (i > 0) {
1382
- const delta = s.reward - prev;
1383
- if (delta < worstDelta) {
1384
- worstDelta = delta;
1385
- worstIdx = i;
1386
- }
1387
- prev = s.reward;
1388
- } else {
1389
- prev = s.reward;
1390
- }
1391
- }
1392
- return {
1393
- runId,
1394
- totalSteps: stepRewards.length,
1395
- meanReward: sumW === 0 ? 0 : sumWR / sumW,
1396
- sumWeightedReward: sumWR,
1397
- failureFraction: failures / stepRewards.length,
1398
- worstStepDelta: worstDelta,
1399
- worstStepIndex: worstIdx
1400
- };
1401
- }
1674
+ if (stepRewards.length === 0) return {
1675
+ runId: "",
1676
+ totalSteps: 0,
1677
+ meanReward: 0,
1678
+ sumWeightedReward: 0,
1679
+ failureFraction: 0,
1680
+ worstStepDelta: 0,
1681
+ worstStepIndex: null
1682
+ };
1683
+ const runId = stepRewards[0].runId;
1684
+ let sumW = 0;
1685
+ let sumWR = 0;
1686
+ let failures = 0;
1687
+ let worstDelta = 0;
1688
+ let worstIdx = null;
1689
+ let prev = stepRewards[0].reward;
1690
+ for (let i = 0; i < stepRewards.length; i++) {
1691
+ const s = stepRewards[i];
1692
+ const w = s.weight ?? 1;
1693
+ sumW += w;
1694
+ sumWR += w * s.reward;
1695
+ if (s.reward < .5) failures++;
1696
+ if (i > 0) {
1697
+ const delta = s.reward - prev;
1698
+ if (delta < worstDelta) {
1699
+ worstDelta = delta;
1700
+ worstIdx = i;
1701
+ }
1702
+ prev = s.reward;
1703
+ } else prev = s.reward;
1704
+ }
1705
+ return {
1706
+ runId,
1707
+ totalSteps: stepRewards.length,
1708
+ meanReward: sumW === 0 ? 0 : sumWR / sumW,
1709
+ sumWeightedReward: sumWR,
1710
+ failureFraction: failures / stepRewards.length,
1711
+ worstStepDelta: worstDelta,
1712
+ worstStepIndex: worstIdx
1713
+ };
1714
+ }
1715
+ /**
1716
+ * Build PRM training triples. The shape: pair runs that share an early
1717
+ * prefix (same scenario, same first N steps) and diverge later — at the
1718
+ * point of divergence, the high-reward run's next step is `chosen`, the
1719
+ * low-reward run's next step is `rejected`. This is the canonical PRM
1720
+ * training data shape from Lightman et al. and DeepSeek-R1 process
1721
+ * supervision.
1722
+ *
1723
+ * Implementation note: we don't have a way to detect "same prefix" in
1724
+ * the general agent setting (token-level prefixes require hashing model
1725
+ * outputs). The current heuristic groups by `(scenarioId, prefixSpanName
1726
+ * sequence)` — runs are paired when their first K span names match. For
1727
+ * production use this should be replaced with a proper trajectory-prefix
1728
+ * hash; the heuristic is good enough for early-stage scaffolding.
1729
+ */
1402
1730
  function prmTrainingPairs(stepRewardsByRun, opts = {}) {
1403
- const minMargin = opts.minMargin ?? 0.2;
1404
- const minPrefix = opts.minPrefixLength ?? 1;
1405
- const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({ runId, steps }));
1406
- const triples = [];
1407
- for (let i = 0; i < runs.length; i++) {
1408
- for (let j = i + 1; j < runs.length; j++) {
1409
- const a = runs[i];
1410
- const b = runs[j];
1411
- const minLen = Math.min(a.steps.length, b.steps.length);
1412
- if (minLen < minPrefix + 1) continue;
1413
- let divergenceIdx = -1;
1414
- for (let k = 0; k < minLen; k++) {
1415
- const sa = a.steps[k];
1416
- const sb = b.steps[k];
1417
- const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name;
1418
- const rewardGap = Math.abs(sa.reward - sb.reward);
1419
- if (structuralDivergence || rewardGap >= minMargin) {
1420
- divergenceIdx = k;
1421
- break;
1422
- }
1423
- }
1424
- if (divergenceIdx < 0) continue;
1425
- if (divergenceIdx < minPrefix) continue;
1426
- const aNext = a.steps[divergenceIdx];
1427
- const bNext = b.steps[divergenceIdx];
1428
- const margin = Math.abs(aNext.reward - bNext.reward);
1429
- if (margin < minMargin) continue;
1430
- const chosen = aNext.reward > bNext.reward ? aNext : bNext;
1431
- const rejected = aNext.reward > bNext.reward ? bNext : aNext;
1432
- const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId;
1433
- const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId;
1434
- triples.push({
1435
- prefixRunId: chosenRun,
1436
- prefixStepIndex: divergenceIdx - 1,
1437
- chosenSpanId: chosen.spanId,
1438
- chosenReward: chosen.reward,
1439
- rejectedSpanId: rejected.spanId,
1440
- rejectedReward: rejected.reward,
1441
- rejectedRunId: rejectedRun,
1442
- marginScore: chosen.reward - rejected.reward
1443
- });
1444
- }
1445
- }
1446
- return triples;
1447
- }
1448
-
1449
- // src/rl/rl-campaign.ts
1731
+ const minMargin = opts.minMargin ?? .2;
1732
+ const minPrefix = opts.minPrefixLength ?? 1;
1733
+ const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({
1734
+ runId,
1735
+ steps
1736
+ }));
1737
+ const triples = [];
1738
+ for (let i = 0; i < runs.length; i++) for (let j = i + 1; j < runs.length; j++) {
1739
+ const a = runs[i];
1740
+ const b = runs[j];
1741
+ const minLen = Math.min(a.steps.length, b.steps.length);
1742
+ if (minLen < minPrefix + 1) continue;
1743
+ let divergenceIdx = -1;
1744
+ for (let k = 0; k < minLen; k++) {
1745
+ const sa = a.steps[k];
1746
+ const sb = b.steps[k];
1747
+ const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name;
1748
+ const rewardGap = Math.abs(sa.reward - sb.reward);
1749
+ if (structuralDivergence || rewardGap >= minMargin) {
1750
+ divergenceIdx = k;
1751
+ break;
1752
+ }
1753
+ }
1754
+ if (divergenceIdx < 0) continue;
1755
+ if (divergenceIdx < minPrefix) continue;
1756
+ const aNext = a.steps[divergenceIdx];
1757
+ const bNext = b.steps[divergenceIdx];
1758
+ if (Math.abs(aNext.reward - bNext.reward) < minMargin) continue;
1759
+ const chosen = aNext.reward > bNext.reward ? aNext : bNext;
1760
+ const rejected = aNext.reward > bNext.reward ? bNext : aNext;
1761
+ const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId;
1762
+ const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId;
1763
+ triples.push({
1764
+ prefixRunId: chosenRun,
1765
+ prefixStepIndex: divergenceIdx - 1,
1766
+ chosenSpanId: chosen.spanId,
1767
+ chosenReward: chosen.reward,
1768
+ rejectedSpanId: rejected.spanId,
1769
+ rejectedReward: rejected.reward,
1770
+ rejectedRunId: rejectedRun,
1771
+ marginScore: chosen.reward - rejected.reward
1772
+ });
1773
+ }
1774
+ return triples;
1775
+ }
1776
+ //#endregion
1777
+ //#region src/rl/rl-campaign.ts
1778
+ /**
1779
+ * `runRLCampaign` — top-level orchestrator that runs the matrix and
1780
+ * produces every RL-ready artifact in one call.
1781
+ *
1782
+ * Wires:
1783
+ * 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)
1784
+ * 2. `extractVerifiableRewardsFromRecords` over the runs, separating deterministic
1785
+ * from probabilistic reward sources for the trainer
1786
+ * 3. `extractPreferences` to produce DPO/PPO/KTO triples
1787
+ * 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)
1788
+ * 5. `rubricPredictiveValidity` against an outcome store, when provided
1789
+ * 6. `detectRewardHacking` as a standing hygiene check
1790
+ * 7. Trainer-format export rows ready for prime-rl / TRL / verl
1791
+ *
1792
+ * The output `RLCampaignResult` is a single, audit-ready artifact: every
1793
+ * stage's output is in there. The consumer's downstream fits in a single
1794
+ * line: pass `result.preferences.pairs` to a DPO trainer,
1795
+ * `result.trainerRows.grpo` to GRPO, or `result.campaign.runs` plus
1796
+ * `result.rewardSignals` to a custom RL loop.
1797
+ */
1450
1798
  async function runRLCampaign(opts) {
1451
- const splitTag = opts.splitTag ?? "search";
1452
- const campaign = await runEvalCampaign({ ...opts, splitTag });
1453
- const rewardSignals = extractVerifiableRewardsFromRecords(
1454
- campaign.runs,
1455
- opts.verifiableReward ?? {}
1456
- );
1457
- const scoredRuns = campaign.runs.filter((run) => runTaskScore(run) !== void 0);
1458
- const { rows: rolloutLines } = await mintRolloutRows(scoredRuns, new InMemoryTraceStore());
1459
- const preferences = extractPreferences(rolloutLines, {
1460
- ...opts.preferences,
1461
- strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
1462
- minMargin: opts.preferences?.minMargin ?? 0.05,
1463
- split: opts.preferences?.split ?? splitTag
1464
- });
1465
- let interimConfidence = null;
1466
- if (opts.report?.comparator) {
1467
- const comparator = opts.report.comparator;
1468
- const deltaSeries = collectPairedDeltaSeries(campaign.runs, comparator);
1469
- if (deltaSeries.some((s) => s.deltas.length > 0)) {
1470
- interimConfidence = evaluateInterimReleaseConfidence({
1471
- deltaSeries,
1472
- alpha: opts.sequential?.alpha,
1473
- bound: opts.sequential?.bound,
1474
- rope: opts.sequential?.rope ?? opts.report?.rope
1475
- });
1476
- }
1477
- }
1478
- const rewardHacking = detectRewardHacking({
1479
- runs: campaign.runs,
1480
- verifiableRewardOptions: opts.verifiableReward
1481
- });
1482
- let predictiveValidity = null;
1483
- if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) {
1484
- predictiveValidity = await rubricPredictiveValidity({
1485
- runs: campaign.runs,
1486
- outcomes: opts.outcomeStore,
1487
- outcomeMetrics: opts.outcomeMetrics
1488
- });
1489
- }
1490
- const trainerRows = {};
1491
- if (opts.trainerExport?.dpo) {
1492
- trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, {
1493
- lines: rolloutLines
1494
- });
1495
- }
1496
- if (opts.trainerExport?.grpo) {
1497
- trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo);
1498
- }
1499
- if (opts.trainerExport?.sft) {
1500
- trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft);
1501
- }
1502
- const summary = buildSummary({
1503
- campaign,
1504
- preferences,
1505
- interimConfidence,
1506
- rewardHacking,
1507
- predictiveValidity
1508
- });
1509
- return {
1510
- campaign,
1511
- rewardSignals,
1512
- preferences,
1513
- interimConfidence,
1514
- rewardHacking,
1515
- predictiveValidity,
1516
- trainerRows,
1517
- summary,
1518
- kind: "agent-eval-rl-campaign"
1519
- };
1799
+ const splitTag = opts.splitTag ?? "search";
1800
+ const campaign = await runEvalCampaign({
1801
+ ...opts,
1802
+ splitTag
1803
+ });
1804
+ const rewardSignals = extractVerifiableRewardsFromRecords(campaign.runs, opts.verifiableReward ?? {});
1805
+ const { rows: rolloutLines } = await mintRolloutRows(campaign.runs.filter((run) => runTaskScore(run) !== void 0), new InMemoryTraceStore());
1806
+ const preferences = extractPreferences(rolloutLines, {
1807
+ ...opts.preferences,
1808
+ strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
1809
+ minMargin: opts.preferences?.minMargin ?? .05,
1810
+ split: opts.preferences?.split ?? splitTag
1811
+ });
1812
+ let interimConfidence = null;
1813
+ if (opts.report?.comparator) {
1814
+ const comparator = opts.report.comparator;
1815
+ const deltaSeries = collectPairedDeltaSeries(campaign.runs, comparator);
1816
+ if (deltaSeries.some((s) => s.deltas.length > 0)) interimConfidence = evaluateInterimReleaseConfidence({
1817
+ deltaSeries,
1818
+ alpha: opts.sequential?.alpha,
1819
+ bound: opts.sequential?.bound,
1820
+ rope: opts.sequential?.rope ?? opts.report?.rope
1821
+ });
1822
+ }
1823
+ const rewardHacking = detectRewardHacking({
1824
+ runs: campaign.runs,
1825
+ verifiableRewardOptions: opts.verifiableReward
1826
+ });
1827
+ let predictiveValidity = null;
1828
+ if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) predictiveValidity = await rubricPredictiveValidity({
1829
+ runs: campaign.runs,
1830
+ outcomes: opts.outcomeStore,
1831
+ outcomeMetrics: opts.outcomeMetrics
1832
+ });
1833
+ const trainerRows = {};
1834
+ if (opts.trainerExport?.dpo) trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, { lines: rolloutLines });
1835
+ if (opts.trainerExport?.grpo) trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo);
1836
+ if (opts.trainerExport?.sft) trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft);
1837
+ const summary = buildSummary({
1838
+ campaign,
1839
+ preferences,
1840
+ interimConfidence,
1841
+ rewardHacking,
1842
+ predictiveValidity
1843
+ });
1844
+ return {
1845
+ campaign,
1846
+ rewardSignals,
1847
+ preferences,
1848
+ interimConfidence,
1849
+ rewardHacking,
1850
+ predictiveValidity,
1851
+ trainerRows,
1852
+ summary,
1853
+ kind: "agent-eval-rl-campaign"
1854
+ };
1520
1855
  }
1521
1856
  function collectPairedDeltaSeries(runs, comparator) {
1522
- const baseline = /* @__PURE__ */ new Map();
1523
- for (const r of runs) {
1524
- if (r.candidateId !== comparator) continue;
1525
- const sid = r.scenarioId;
1526
- const score = runTaskScore(r);
1527
- if (score === void 0) continue;
1528
- baseline.set(`${sid}::${r.seed}`, score);
1529
- }
1530
- const byCandidate = /* @__PURE__ */ new Map();
1531
- for (const r of runs) {
1532
- if (r.candidateId === comparator) continue;
1533
- const sid = r.scenarioId;
1534
- const score = runTaskScore(r);
1535
- if (score === void 0) continue;
1536
- const baseScore = baseline.get(`${sid}::${r.seed}`);
1537
- if (typeof baseScore !== "number") continue;
1538
- const arr = byCandidate.get(r.candidateId) ?? [];
1539
- arr.push(score - baseScore);
1540
- byCandidate.set(r.candidateId, arr);
1541
- }
1542
- return [...byCandidate.entries()].map(([candidateId, deltas]) => ({ candidateId, deltas }));
1857
+ const baseline = /* @__PURE__ */ new Map();
1858
+ for (const r of runs) {
1859
+ if (r.candidateId !== comparator) continue;
1860
+ const sid = r.scenarioId;
1861
+ const score = runTaskScore(r);
1862
+ if (score === void 0) continue;
1863
+ baseline.set(`${sid}::${r.seed}`, score);
1864
+ }
1865
+ const byCandidate = /* @__PURE__ */ new Map();
1866
+ for (const r of runs) {
1867
+ if (r.candidateId === comparator) continue;
1868
+ const sid = r.scenarioId;
1869
+ const score = runTaskScore(r);
1870
+ if (score === void 0) continue;
1871
+ const baseScore = baseline.get(`${sid}::${r.seed}`);
1872
+ if (typeof baseScore !== "number") continue;
1873
+ const arr = byCandidate.get(r.candidateId) ?? [];
1874
+ arr.push(score - baseScore);
1875
+ byCandidate.set(r.candidateId, arr);
1876
+ }
1877
+ return [...byCandidate.entries()].map(([candidateId, deltas]) => ({
1878
+ candidateId,
1879
+ deltas
1880
+ }));
1543
1881
  }
1544
1882
  function buildSummary(args) {
1545
- const c = args.campaign;
1546
- const lines = [
1547
- `${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}\u2026)`,
1548
- `preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`
1549
- ];
1550
- if (args.interimConfidence) {
1551
- lines.push(
1552
- `sequential verdict: ${args.interimConfidence.recommendation.decision}` + (args.interimConfidence.recommendation.candidateId ? ` ${args.interimConfidence.recommendation.candidateId}` : "")
1553
- );
1554
- }
1555
- lines.push(
1556
- `reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`
1557
- );
1558
- if (args.predictiveValidity) {
1559
- const top = args.predictiveValidity.ranked[0];
1560
- lines.push(
1561
- `top-rubric: ${top?.rubric ?? "none"} \u03C1=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? "no data"})`
1562
- );
1563
- }
1564
- return lines.join(" | ");
1565
- }
1566
-
1567
- // src/rl/run-record-adapters.ts
1883
+ const c = args.campaign;
1884
+ const lines = [`${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}…)`, `preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`];
1885
+ if (args.interimConfidence) lines.push(`sequential verdict: ${args.interimConfidence.recommendation.decision}` + (args.interimConfidence.recommendation.candidateId ? ` ${args.interimConfidence.recommendation.candidateId}` : ""));
1886
+ lines.push(`reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`);
1887
+ if (args.predictiveValidity) {
1888
+ const top = args.predictiveValidity.ranked[0];
1889
+ lines.push(`top-rubric: ${top?.rubric ?? "none"} ρ=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? "no data"})`);
1890
+ }
1891
+ return lines.join(" | ");
1892
+ }
1893
+ //#endregion
1894
+ //#region src/rl/run-record-adapters.ts
1895
+ /**
1896
+ * Adapters: convert measurement outputs into the canonical `RunRecord[]`
1897
+ * artifact that `replayCache`, `pairedEvalueSequence`, and
1898
+ * `rubricPredictiveValidity` consume. Two sources:
1899
+ * - `campaignToRunRecords` the campaign substrate's per-cell results
1900
+ * (the modern path: `runCampaign` / `runImprovementLoop` → records).
1901
+ * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
1902
+ *
1903
+ * Adapters are thin and explicit — every mandatory `RunRecord` field comes
1904
+ * from a caller-supplied context (`commitSha`, `model`, `promptHash`,
1905
+ * `configHash`) plus the cell's runtime data. The validator still rejects
1906
+ * bare-alias model strings — the caller snapshot-pins.
1907
+ */
1908
+ /**
1909
+ * Convert a `CampaignResult` into canonical `RunRecord[]`, one per cell.
1910
+ * Successful judged cells carry their mean judge composite and dimensions.
1911
+ * Errored or unjudged cells remain unlabeled while retaining explicit terminal
1912
+ * outcome, execution-error count, token usage, cost, and failure detail.
1913
+ * `candidateId` identifies the measured surface and defaults to the campaign
1914
+ * manifest hash.
1915
+ */
1568
1916
  function campaignToRunRecords(campaign, ctx) {
1569
- const splitTag = ctx.splitTag ?? "search";
1570
- const candidateId = ctx.candidateId ?? campaign.manifestHash;
1571
- return campaign.cells.map(
1572
- (cell) => campaignCellToRunRecord(cell, {
1573
- runId: cell.cellId,
1574
- experimentId: ctx.experimentId,
1575
- candidateId,
1576
- model: ctx.model,
1577
- promptHash: ctx.promptHash,
1578
- configHash: ctx.configHash,
1579
- commitSha: ctx.commitSha,
1580
- splitTag,
1581
- defaultCostUsd: ctx.defaultCostUsd
1582
- })
1583
- );
1584
- }
1917
+ const splitTag = ctx.splitTag ?? "search";
1918
+ const candidateId = ctx.candidateId ?? campaign.manifestHash;
1919
+ return campaign.cells.map((cell) => campaignCellToRunRecord(cell, {
1920
+ runId: cell.cellId,
1921
+ experimentId: ctx.experimentId,
1922
+ candidateId,
1923
+ model: ctx.model,
1924
+ promptHash: ctx.promptHash,
1925
+ configHash: ctx.configHash,
1926
+ commitSha: ctx.commitSha,
1927
+ splitTag,
1928
+ defaultCostUsd: ctx.defaultCostUsd
1929
+ }));
1930
+ }
1931
+ /**
1932
+ * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
1933
+ * A split score is emitted only when `report.taskScore` proves the configured
1934
+ * scoring panel completed. Partial scores remain in `outcome.raw` for
1935
+ * diagnosis. Layer errors and timeouts become judge or execution telemetry;
1936
+ * only a scored `fail` layer may produce task-failure detail.
1937
+ */
1585
1938
  function verificationReportToRunRecord(report, ctx, opts = {}) {
1586
- const splitTag = ctx.splitTag ?? "search";
1587
- const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
1588
- const hasValidLayerMeasurement = report.layers.some(hasValidTaskMeasurement);
1589
- const taskScore = hasValidLayerMeasurement && isValidScore(report.taskScore) ? report.taskScore : void 0;
1590
- let executionErrorCount = 0;
1591
- let judgeErrorCount = 0;
1592
- let layerErrorCount = 0;
1593
- let layerTimeoutCount = 0;
1594
- let unscoredLayerCount = 0;
1595
- const raw = {
1596
- pass_count: report.passCount,
1597
- fail_count: report.failCount,
1598
- error_count: report.errorCount,
1599
- skipped_count: report.skippedCount,
1600
- duration_ms: report.durationMs,
1601
- execution_error_count: 0
1602
- };
1603
- for (const layer of report.layers) {
1604
- if (hasValidTaskMeasurement(layer)) raw[`layer.${layer.layer}`] = layer.score;
1605
- else unscoredLayerCount++;
1606
- raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
1607
- if (layer.status === "error" || layer.status === "timeout") {
1608
- if (layer.errorSource === "judge") judgeErrorCount++;
1609
- else executionErrorCount++;
1610
- if (layer.status === "error") layerErrorCount++;
1611
- else layerTimeoutCount++;
1612
- }
1613
- if (layer.diagnostics) {
1614
- for (const [k, v] of Object.entries(layer.diagnostics)) {
1615
- if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
1616
- }
1617
- }
1618
- }
1619
- raw.execution_error_count = executionErrorCount;
1620
- if (judgeErrorCount > 0) raw.judge_error_count = judgeErrorCount;
1621
- if (layerErrorCount > 0) raw.layer_error_count = layerErrorCount;
1622
- if (layerTimeoutCount > 0) raw.layer_timeout_count = layerTimeoutCount;
1623
- if (unscoredLayerCount > 0) raw.unscored_layer_count = unscoredLayerCount;
1624
- if (taskScore !== void 0) raw.blended_score = taskScore;
1625
- const firstScoredFailure = report.layers.find(
1626
- (layer) => layer.status === "fail" && hasValidTaskMeasurement(layer)
1627
- );
1628
- const outcome = { raw };
1629
- if (taskScore !== void 0) {
1630
- if (splitTag === "holdout") outcome.holdoutScore = taskScore;
1631
- else outcome.searchScore = taskScore;
1632
- }
1633
- return {
1634
- runId,
1635
- experimentId: ctx.experimentId,
1636
- candidateId: ctx.candidateId,
1637
- seed: 0,
1638
- model: ctx.model,
1639
- promptHash: ctx.promptHash,
1640
- configHash: ctx.configHash,
1641
- commitSha: ctx.commitSha,
1642
- wallMs: report.durationMs,
1643
- costUsd: ctx.defaultCostUsd ?? null,
1644
- costProvenance: ctx.defaultCostUsd === void 0 ? { kind: "uncaptured", usd: null } : { kind: "estimated", usd: ctx.defaultCostUsd },
1645
- tokenUsage: { input: 0, output: 0 },
1646
- terminalOutcome: "succeeded",
1647
- outcome,
1648
- ...firstScoredFailure ? {
1649
- failureClass: "unknown",
1650
- failureMode: `layer_${firstScoredFailure.layer}_fail`
1651
- } : {},
1652
- splitTag,
1653
- scenarioId: ctx.scenarioId
1654
- };
1939
+ const splitTag = ctx.splitTag ?? "search";
1940
+ const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
1941
+ const taskScore = report.layers.some(hasValidTaskMeasurement) && isValidScore(report.taskScore) ? report.taskScore : void 0;
1942
+ let executionErrorCount = 0;
1943
+ let judgeErrorCount = 0;
1944
+ let layerErrorCount = 0;
1945
+ let layerTimeoutCount = 0;
1946
+ let unscoredLayerCount = 0;
1947
+ const raw = {
1948
+ pass_count: report.passCount,
1949
+ fail_count: report.failCount,
1950
+ error_count: report.errorCount,
1951
+ skipped_count: report.skippedCount,
1952
+ duration_ms: report.durationMs,
1953
+ execution_error_count: 0
1954
+ };
1955
+ for (const layer of report.layers) {
1956
+ if (hasValidTaskMeasurement(layer)) raw[`layer.${layer.layer}`] = layer.score;
1957
+ else unscoredLayerCount++;
1958
+ raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
1959
+ if (layer.status === "error" || layer.status === "timeout") {
1960
+ if (layer.errorSource === "judge") judgeErrorCount++;
1961
+ else executionErrorCount++;
1962
+ if (layer.status === "error") layerErrorCount++;
1963
+ else layerTimeoutCount++;
1964
+ }
1965
+ if (layer.diagnostics) {
1966
+ for (const [k, v] of Object.entries(layer.diagnostics)) if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
1967
+ }
1968
+ }
1969
+ raw.execution_error_count = executionErrorCount;
1970
+ if (judgeErrorCount > 0) raw.judge_error_count = judgeErrorCount;
1971
+ if (layerErrorCount > 0) raw.layer_error_count = layerErrorCount;
1972
+ if (layerTimeoutCount > 0) raw.layer_timeout_count = layerTimeoutCount;
1973
+ if (unscoredLayerCount > 0) raw.unscored_layer_count = unscoredLayerCount;
1974
+ if (taskScore !== void 0) raw.blended_score = taskScore;
1975
+ const firstScoredFailure = report.layers.find((layer) => layer.status === "fail" && hasValidTaskMeasurement(layer));
1976
+ const outcome = { raw };
1977
+ if (taskScore !== void 0) if (splitTag === "holdout") outcome.holdoutScore = taskScore;
1978
+ else outcome.searchScore = taskScore;
1979
+ return {
1980
+ runId,
1981
+ experimentId: ctx.experimentId,
1982
+ candidateId: ctx.candidateId,
1983
+ seed: 0,
1984
+ model: ctx.model,
1985
+ promptHash: ctx.promptHash,
1986
+ configHash: ctx.configHash,
1987
+ commitSha: ctx.commitSha,
1988
+ wallMs: report.durationMs,
1989
+ costUsd: ctx.defaultCostUsd ?? null,
1990
+ costProvenance: ctx.defaultCostUsd === void 0 ? {
1991
+ kind: "uncaptured",
1992
+ usd: null
1993
+ } : {
1994
+ kind: "estimated",
1995
+ usd: ctx.defaultCostUsd
1996
+ },
1997
+ tokenUsage: {
1998
+ input: 0,
1999
+ output: 0
2000
+ },
2001
+ terminalOutcome: "succeeded",
2002
+ outcome,
2003
+ ...firstScoredFailure ? {
2004
+ failureClass: "unknown",
2005
+ failureMode: `layer_${firstScoredFailure.layer}_fail`
2006
+ } : {},
2007
+ splitTag,
2008
+ scenarioId: ctx.scenarioId
2009
+ };
1655
2010
  }
1656
2011
  function hasValidTaskMeasurement(layer) {
1657
- return (layer.status === "pass" || layer.status === "fail") && isValidScore(layer.score);
2012
+ return (layer.status === "pass" || layer.status === "fail") && isValidScore(layer.score);
1658
2013
  }
1659
2014
  function isValidScore(score) {
1660
- return typeof score === "number" && Number.isFinite(score) && score >= 0 && score <= 1;
1661
- }
1662
-
1663
- // src/rl/sim-fidelity.ts
1664
- var ABSENT_CATEGORY = "(absent)";
1665
- var DEFAULT_MIN_N_PER_FEATURE = 20;
1666
- var DEFAULT_QUANTILE_BUCKETS = 4;
1667
- var REPRESENTATIVE_MIN_FIDELITY = 0.8;
1668
- var TOP_SHIFT_COUNT = 5;
1669
- var defaultBehaviorFeatures = (record) => {
1670
- const raw = record.outcome?.raw ?? {};
1671
- const toolErrors = finiteOrNull(raw.tool_errors);
1672
- const turnsAborted = finiteOrNull(raw.turns_aborted);
1673
- const completion = record.completion;
1674
- return {
1675
- // RAW (`observedSplitScore`), deliberately: this feature vector is one
1676
- // half of a sim-vs-production divergence measurement. Gating a gamed run to
1677
- // 0 would move the simulated distribution toward production and report the
1678
- // simulator as MORE faithful precisely where it is being gamed. Each split
1679
- // is read separately rather than through `observedScore` so a non-finite
1680
- // holdout score falls back to search instead of poisoning the bucket.
1681
- score: finiteOrNull(observedSplitScore(record, "holdout")) ?? finiteOrNull(observedSplitScore(record, "search")),
1682
- failure_class: record.failureClass ?? null,
1683
- wall_ms: finiteOrNull(record.wallMs),
1684
- output_tokens: finiteOrNull(record.tokenUsage?.output),
1685
- turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),
1686
- tool_errors: toolErrors,
1687
- tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),
1688
- completion_length: typeof completion === "string" ? completion.length : null
1689
- };
2015
+ return typeof score === "number" && Number.isFinite(score) && score >= 0 && score <= 1;
2016
+ }
2017
+ //#endregion
2018
+ //#region src/rl/sim-fidelity.ts
2019
+ /**
2020
+ * Simulator fidelity — score a user SIMULATOR's realism against real-user
2021
+ * trace distributions.
2022
+ *
2023
+ * Synthetic-persona evals (`PersonaConfig`-driven canonical evals, fuzz
2024
+ * user-simulator objectives) stand in for real users in most of the numbers
2025
+ * we publish. The standing threat is the Sim2Real gap: a simulator that is
2026
+ * distributionally unlike production creates "easy mode" and silently
2027
+ * inflates every score built on it. This module measures that gap from the
2028
+ * SAME artifact both sides already produce — `RunRecord`s — so no new
2029
+ * capture pipeline is needed:
2030
+ *
2031
+ * - `simFidelityReport` per-feature Jensen-Shannon divergence between
2032
+ * simulated and production record distributions, collapsed into a
2033
+ * fidelity coefficient in [0,1].
2034
+ * - `easyModeCheck` the headline academic failure mode (sim inflates
2035
+ * pass-rate over production) as its own named artifact.
2036
+ *
2037
+ * Every synthetic-persona eval result should publish its fidelity
2038
+ * coefficient alongside the score — a number from an unrepresentative
2039
+ * simulator is an unlabeled estimate. Wire-in points:
2040
+ *
2041
+ * - canonical persona evals: pass the campaign's `RunRecord`s as
2042
+ * `simulated` and intake-adapter output (`contract/intake`: OTel spans,
2043
+ * feedback tables, coding-agent sessions) as `production`
2044
+ * - the fuzz user-sim objective: use `1 - report.fidelity` as a realism
2045
+ * penalty when searching over generated personas
2046
+ * - the durable corpus (`./corpus`): both sides read straight from
2047
+ * `readCorpus` — tag sim vs production by `experimentId`
2048
+ */
2049
+ /** Reserved histogram category for `null` feature values. A capture-rate
2050
+ * difference (one side instruments a signal, the other does not) registers
2051
+ * as divergence by design: a simulator that produces no tool traces is not
2052
+ * representative of production that does. */
2053
+ const ABSENT_CATEGORY = "(absent)";
2054
+ /** Minimum non-null observations PER SIDE for a feature to enter the
2055
+ * fidelity mean. Below this the JSD estimate is sampling noise. */
2056
+ const DEFAULT_MIN_N_PER_FEATURE = 20;
2057
+ /** Quantile buckets used to discretize numeric features. Quartiles balance
2058
+ * resolution against per-bucket sample size at the default minN. */
2059
+ const DEFAULT_QUANTILE_BUCKETS = 4;
2060
+ /** Fidelity at or above this → 'representative'; below → 'skewed'.
2061
+ * 1 − 0.8 = mean JSD 0.2 ≈ distributions that mostly overlap with one
2062
+ * clearly shifted mode — the point where per-feature shifts start changing
2063
+ * which failure classes an eval can even observe. */
2064
+ const REPRESENTATIVE_MIN_FIDELITY = .8;
2065
+ const TOP_SHIFT_COUNT = 5;
2066
+ /**
2067
+ * Default feature set — ONLY fields verified present on both simulated and
2068
+ * production records:
2069
+ *
2070
+ * - `score`, `wall_ms`, `output_tokens` — mandatory per the `RunRecord`
2071
+ * validator (non-finite values read as absent rather than poisoning a
2072
+ * bucket).
2073
+ * - `failure_class` — optional taxonomy field; absent counted explicitly.
2074
+ * - `turn_count`, `tool_errors`, `tool_error_recovery` — derived from the
2075
+ * `outcome.raw` counters the intake adapters and eval harnesses write
2076
+ * (`turns_completed`, `assistant_messages`, `tool_errors`,
2077
+ * `turns_aborted`); absent on records whose producer did not capture
2078
+ * them, counted explicitly.
2079
+ * - `completion_length` — from the optional `CorpusRecord` trajectory
2080
+ * text; the message-length proxy when records come from the corpus.
2081
+ *
2082
+ * `RunRecord` carries event COUNTS, not event ordering, so
2083
+ * `tool_error_recovery` is a counts-only derivation: errors occurred and the
2084
+ * run still completed cleanly ('recovered') vs aborted or classified as a
2085
+ * failure ('unrecovered') — not a literal error→retry sequence check.
2086
+ */
2087
+ const defaultBehaviorFeatures = (record) => {
2088
+ const raw = record.outcome?.raw ?? {};
2089
+ const toolErrors = finiteOrNull(raw.tool_errors);
2090
+ const turnsAborted = finiteOrNull(raw.turns_aborted);
2091
+ const completion = record.completion;
2092
+ return {
2093
+ score: finiteOrNull(observedSplitScore(record, "holdout")) ?? finiteOrNull(observedSplitScore(record, "search")),
2094
+ failure_class: record.failureClass ?? null,
2095
+ wall_ms: finiteOrNull(record.wallMs),
2096
+ output_tokens: finiteOrNull(record.tokenUsage?.output),
2097
+ turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),
2098
+ tool_errors: toolErrors,
2099
+ tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),
2100
+ completion_length: typeof completion === "string" ? completion.length : null
2101
+ };
1690
2102
  };
1691
2103
  function toolErrorRecovery(toolErrors, turnsAborted, failureClass) {
1692
- if (toolErrors === null) return null;
1693
- if (toolErrors === 0) return "no-tool-errors";
1694
- const failed = (turnsAborted ?? 0) > 0 || failureClass !== void 0 && failureClass !== "success";
1695
- return failed ? "unrecovered" : "recovered";
2104
+ if (toolErrors === null) return null;
2105
+ if (toolErrors === 0) return "no-tool-errors";
2106
+ return (turnsAborted ?? 0) > 0 || failureClass !== void 0 && failureClass !== "success" ? "unrecovered" : "recovered";
1696
2107
  }
1697
2108
  function finiteOrNull(value) {
1698
- return typeof value === "number" && Number.isFinite(value) ? value : null;
1699
- }
2109
+ return typeof value === "number" && Number.isFinite(value) ? value : null;
2110
+ }
2111
+ /**
2112
+ * Jensen-Shannon divergence between two categorical histograms (raw counts;
2113
+ * normalized internally). Log base 2 → bounded [0,1]: 0 = identical
2114
+ * distributions, 1 = disjoint support. Symmetric, defined even where the
2115
+ * supports differ — exactly the regime sim-vs-production comparison lives in.
2116
+ * Throws on zero-mass or negative/non-finite counts: an empty histogram has
2117
+ * no distribution and a silent 0 would read as "perfectly representative".
2118
+ */
1700
2119
  function jsDivergence(p, q) {
1701
- const keys = /* @__PURE__ */ new Set([...Object.keys(p), ...Object.keys(q)]);
1702
- if (keys.size === 0) {
1703
- throw new ValidationError("jsDivergence: both histograms are empty");
1704
- }
1705
- let pSum = 0;
1706
- let qSum = 0;
1707
- for (const key of keys) {
1708
- const pv = p[key] ?? 0;
1709
- const qv = q[key] ?? 0;
1710
- if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) {
1711
- throw new ValidationError(`jsDivergence: negative or non-finite count for category "${key}"`);
1712
- }
1713
- pSum += pv;
1714
- qSum += qv;
1715
- }
1716
- if (pSum === 0 || qSum === 0) {
1717
- throw new ValidationError("jsDivergence: a histogram with zero total mass has no distribution");
1718
- }
1719
- let divergence = 0;
1720
- for (const key of keys) {
1721
- const pp = (p[key] ?? 0) / pSum;
1722
- const qp = (q[key] ?? 0) / qSum;
1723
- const m = (pp + qp) / 2;
1724
- if (pp > 0) divergence += 0.5 * pp * Math.log2(pp / m);
1725
- if (qp > 0) divergence += 0.5 * qp * Math.log2(qp / m);
1726
- }
1727
- return Math.min(1, Math.max(0, divergence));
1728
- }
1729
- function quantileEdges(values, bucketCount = DEFAULT_QUANTILE_BUCKETS) {
1730
- if (values.length === 0) {
1731
- throw new ValidationError("quantileEdges: requires at least one value");
1732
- }
1733
- if (!Number.isInteger(bucketCount) || bucketCount < 2) {
1734
- throw new ValidationError(
1735
- `quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`
1736
- );
1737
- }
1738
- const sorted = [...values].sort((a, b) => a - b);
1739
- const edges = [];
1740
- for (let k = 1; k < bucketCount; k++) {
1741
- const pos = k / bucketCount * (sorted.length - 1);
1742
- const lo = sorted[Math.floor(pos)];
1743
- const hi = sorted[Math.ceil(pos)];
1744
- edges.push(lo + (pos - Math.floor(pos)) * (hi - lo));
1745
- }
1746
- return [...new Set(edges)];
1747
- }
2120
+ const keys = /* @__PURE__ */ new Set([...Object.keys(p), ...Object.keys(q)]);
2121
+ if (keys.size === 0) throw new ValidationError("jsDivergence: both histograms are empty");
2122
+ let pSum = 0;
2123
+ let qSum = 0;
2124
+ for (const key of keys) {
2125
+ const pv = p[key] ?? 0;
2126
+ const qv = q[key] ?? 0;
2127
+ if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) throw new ValidationError(`jsDivergence: negative or non-finite count for category "${key}"`);
2128
+ pSum += pv;
2129
+ qSum += qv;
2130
+ }
2131
+ if (pSum === 0 || qSum === 0) throw new ValidationError("jsDivergence: a histogram with zero total mass has no distribution");
2132
+ let divergence = 0;
2133
+ for (const key of keys) {
2134
+ const pp = (p[key] ?? 0) / pSum;
2135
+ const qp = (q[key] ?? 0) / qSum;
2136
+ const m = (pp + qp) / 2;
2137
+ if (pp > 0) divergence += .5 * pp * Math.log2(pp / m);
2138
+ if (qp > 0) divergence += .5 * qp * Math.log2(qp / m);
2139
+ }
2140
+ return Math.min(1, Math.max(0, divergence));
2141
+ }
2142
+ /**
2143
+ * Deterministic quantile edges over a value set (the UNION of both sides, so
2144
+ * sim and production land in the same buckets). Linear interpolation between
2145
+ * order statistics; duplicate edges from heavy ties collapse into fewer,
2146
+ * wider buckets. Returns `bucketCount - 1` edges before deduplication.
2147
+ */
2148
+ function quantileEdges(values, bucketCount = 4) {
2149
+ if (values.length === 0) throw new ValidationError("quantileEdges: requires at least one value");
2150
+ if (!Number.isInteger(bucketCount) || bucketCount < 2) throw new ValidationError(`quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`);
2151
+ const sorted = [...values].sort((a, b) => a - b);
2152
+ const edges = [];
2153
+ for (let k = 1; k < bucketCount; k++) {
2154
+ const pos = k / bucketCount * (sorted.length - 1);
2155
+ const lo = sorted[Math.floor(pos)];
2156
+ const hi = sorted[Math.ceil(pos)];
2157
+ edges.push(lo + (pos - Math.floor(pos)) * (hi - lo));
2158
+ }
2159
+ return [...new Set(edges)];
2160
+ }
2161
+ /** Stable half-open bucket label for a value against quantile edges:
2162
+ * `[-inf,e0)`, `[e0,e1)`, …, `[eLast,+inf)`. */
1748
2163
  function bucketLabel(value, edges) {
1749
- let i = 0;
1750
- while (i < edges.length && value >= edges[i]) i++;
1751
- const lo = i === 0 ? "-inf" : String(edges[i - 1]);
1752
- const hi = i === edges.length ? "+inf" : String(edges[i]);
1753
- return `[${lo},${hi})`;
1754
- }
2164
+ let i = 0;
2165
+ while (i < edges.length && value >= edges[i]) i++;
2166
+ return `[${i === 0 ? "-inf" : String(edges[i - 1])},${i === edges.length ? "+inf" : String(edges[i])})`;
2167
+ }
2168
+ /**
2169
+ * Compare a simulator's RunRecords against production RunRecords, feature by
2170
+ * feature. Numeric features are bucketed by deterministic quantiles of the
2171
+ * union; nulls count as an explicit `ABSENT_CATEGORY`. Throws on empty
2172
+ * inputs — "no records" is a wiring error, not a distribution.
2173
+ */
1755
2174
  function simFidelityReport(simulated, production, opts = {}) {
1756
- if (simulated.length === 0) {
1757
- throw new ValidationError("simFidelityReport: simulated records are empty");
1758
- }
1759
- if (production.length === 0) {
1760
- throw new ValidationError("simFidelityReport: production records are empty");
1761
- }
1762
- const extract = opts.features ?? defaultBehaviorFeatures;
1763
- const minN = opts.minNPerFeature ?? DEFAULT_MIN_N_PER_FEATURE;
1764
- const simMaps = simulated.map(extract);
1765
- const prodMaps = production.map(extract);
1766
- const featureNames = [];
1767
- const seen = /* @__PURE__ */ new Set();
1768
- for (const map of [...simMaps, ...prodMaps]) {
1769
- for (const name of Object.keys(map)) {
1770
- if (!seen.has(name)) {
1771
- seen.add(name);
1772
- featureNames.push(name);
1773
- }
1774
- }
1775
- }
1776
- const perDimension = [];
1777
- const insufficientData = [];
1778
- for (const feature of featureNames) {
1779
- const simVals = simMaps.map((m) => m[feature] ?? null);
1780
- const prodVals = prodMaps.map((m) => m[feature] ?? null);
1781
- const nSim = simVals.filter((v) => v !== null).length;
1782
- const nProd = prodVals.filter((v) => v !== null).length;
1783
- if (nSim < minN || nProd < minN) {
1784
- insufficientData.push(feature);
1785
- continue;
1786
- }
1787
- const { sim, prod } = histograms(feature, simVals, prodVals);
1788
- perDimension.push({
1789
- feature,
1790
- divergence: jsDivergence(sim, prod),
1791
- topShifts: topShifts(sim, simVals.length, prod, prodVals.length),
1792
- nSim,
1793
- nProd
1794
- });
1795
- }
1796
- if (perDimension.length === 0) {
1797
- return { perDimension, fidelity: Number.NaN, insufficientData, verdict: "insufficient-data" };
1798
- }
1799
- const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length;
1800
- return {
1801
- perDimension,
1802
- fidelity,
1803
- insufficientData,
1804
- verdict: fidelity >= REPRESENTATIVE_MIN_FIDELITY ? "representative" : "skewed"
1805
- };
2175
+ if (simulated.length === 0) throw new ValidationError("simFidelityReport: simulated records are empty");
2176
+ if (production.length === 0) throw new ValidationError("simFidelityReport: production records are empty");
2177
+ const extract = opts.features ?? defaultBehaviorFeatures;
2178
+ const minN = opts.minNPerFeature ?? 20;
2179
+ const simMaps = simulated.map(extract);
2180
+ const prodMaps = production.map(extract);
2181
+ const featureNames = [];
2182
+ const seen = /* @__PURE__ */ new Set();
2183
+ for (const map of [...simMaps, ...prodMaps]) for (const name of Object.keys(map)) if (!seen.has(name)) {
2184
+ seen.add(name);
2185
+ featureNames.push(name);
2186
+ }
2187
+ const perDimension = [];
2188
+ const insufficientData = [];
2189
+ for (const feature of featureNames) {
2190
+ const simVals = simMaps.map((m) => m[feature] ?? null);
2191
+ const prodVals = prodMaps.map((m) => m[feature] ?? null);
2192
+ const nSim = simVals.filter((v) => v !== null).length;
2193
+ const nProd = prodVals.filter((v) => v !== null).length;
2194
+ if (nSim < minN || nProd < minN) {
2195
+ insufficientData.push(feature);
2196
+ continue;
2197
+ }
2198
+ const { sim, prod } = histograms(feature, simVals, prodVals);
2199
+ perDimension.push({
2200
+ feature,
2201
+ divergence: jsDivergence(sim, prod),
2202
+ topShifts: topShifts(sim, simVals.length, prod, prodVals.length),
2203
+ nSim,
2204
+ nProd
2205
+ });
2206
+ }
2207
+ if (perDimension.length === 0) return {
2208
+ perDimension,
2209
+ fidelity: NaN,
2210
+ insufficientData,
2211
+ verdict: "insufficient-data"
2212
+ };
2213
+ const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length;
2214
+ return {
2215
+ perDimension,
2216
+ fidelity,
2217
+ insufficientData,
2218
+ verdict: fidelity >= .8 ? "representative" : "skewed"
2219
+ };
1806
2220
  }
1807
2221
  function histograms(feature, simVals, prodVals) {
1808
- const kinds = /* @__PURE__ */ new Set();
1809
- for (const v of [...simVals, ...prodVals]) {
1810
- if (v !== null) kinds.add(typeof v);
1811
- }
1812
- if (kinds.size > 1) {
1813
- throw new ValidationError(
1814
- `simFidelityReport: feature "${feature}" mixes string and number values \u2014 an extractor must return one kind per feature`
1815
- );
1816
- }
1817
- let toCategory;
1818
- if (kinds.has("number")) {
1819
- const union = [];
1820
- for (const v of [...simVals, ...prodVals]) {
1821
- if (v !== null) union.push(v);
1822
- }
1823
- const edges = quantileEdges(union);
1824
- toCategory = (v) => bucketLabel(v, edges);
1825
- } else {
1826
- toCategory = (v) => v;
1827
- }
1828
- const count = (vals) => {
1829
- const hist = {};
1830
- for (const v of vals) {
1831
- const key = v === null ? ABSENT_CATEGORY : toCategory(v);
1832
- hist[key] = (hist[key] ?? 0) + 1;
1833
- }
1834
- return hist;
1835
- };
1836
- return { sim: count(simVals), prod: count(prodVals) };
2222
+ const kinds = /* @__PURE__ */ new Set();
2223
+ for (const v of [...simVals, ...prodVals]) if (v !== null) kinds.add(typeof v);
2224
+ if (kinds.size > 1) throw new ValidationError(`simFidelityReport: feature "${feature}" mixes string and number values — an extractor must return one kind per feature`);
2225
+ let toCategory;
2226
+ if (kinds.has("number")) {
2227
+ const union = [];
2228
+ for (const v of [...simVals, ...prodVals]) if (v !== null) union.push(v);
2229
+ const edges = quantileEdges(union);
2230
+ toCategory = (v) => bucketLabel(v, edges);
2231
+ } else toCategory = (v) => v;
2232
+ const count = (vals) => {
2233
+ const hist = {};
2234
+ for (const v of vals) {
2235
+ const key = v === null ? ABSENT_CATEGORY : toCategory(v);
2236
+ hist[key] = (hist[key] ?? 0) + 1;
2237
+ }
2238
+ return hist;
2239
+ };
2240
+ return {
2241
+ sim: count(simVals),
2242
+ prod: count(prodVals)
2243
+ };
1837
2244
  }
1838
2245
  function topShifts(sim, simTotal, prod, prodTotal) {
1839
- const keys = [.../* @__PURE__ */ new Set([...Object.keys(sim), ...Object.keys(prod)])];
1840
- const shifts = keys.map((value) => ({
1841
- value,
1842
- pSim: (sim[value] ?? 0) / simTotal,
1843
- pProd: (prod[value] ?? 0) / prodTotal
1844
- }));
1845
- shifts.sort((a, b) => {
1846
- const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd);
1847
- return delta !== 0 ? delta : a.value.localeCompare(b.value);
1848
- });
1849
- return shifts.slice(0, TOP_SHIFT_COUNT);
1850
- }
2246
+ const shifts = [.../* @__PURE__ */ new Set([...Object.keys(sim), ...Object.keys(prod)])].map((value) => ({
2247
+ value,
2248
+ pSim: (sim[value] ?? 0) / simTotal,
2249
+ pProd: (prod[value] ?? 0) / prodTotal
2250
+ }));
2251
+ shifts.sort((a, b) => {
2252
+ const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd);
2253
+ return delta !== 0 ? delta : a.value.localeCompare(b.value);
2254
+ });
2255
+ return shifts.slice(0, TOP_SHIFT_COUNT);
2256
+ }
2257
+ /**
2258
+ * The headline simulator failure mode as its own named artifact: a simulator
2259
+ * that creates "easy mode" inflates pass-rate relative to production, and
2260
+ * every score measured against it overstates reality. Throws on empty inputs
2261
+ * and on records carrying neither score — a silently-skipped record would
2262
+ * bias the very rate this check exists to keep honest.
2263
+ */
1851
2264
  function easyModeCheck(simulated, production, opts = {}) {
1852
- if (simulated.length === 0) {
1853
- throw new ValidationError("easyModeCheck: simulated records are empty");
1854
- }
1855
- if (production.length === 0) {
1856
- throw new ValidationError("easyModeCheck: production records are empty");
1857
- }
1858
- const threshold = opts.passThreshold ?? 0.5;
1859
- const tolerance = opts.inflationTolerance ?? 0.1;
1860
- const passRate = (records, side) => {
1861
- let passes = 0;
1862
- for (const r of records) {
1863
- const score = finiteOrNull(observedSplitScore(r, "holdout")) ?? finiteOrNull(observedSplitScore(r, "search"));
1864
- if (score === null) {
1865
- throw new ValidationError(
1866
- `easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`
1867
- );
1868
- }
1869
- if (score >= threshold) passes++;
1870
- }
1871
- return passes / records.length;
1872
- };
1873
- const simPassRate = passRate(simulated, "simulated");
1874
- const prodPassRate = passRate(production, "production");
1875
- const gap = simPassRate - prodPassRate;
1876
- return { simPassRate, prodPassRate, gap, inflated: gap > tolerance };
1877
- }
1878
-
1879
- // src/rl/tournament.ts
2265
+ if (simulated.length === 0) throw new ValidationError("easyModeCheck: simulated records are empty");
2266
+ if (production.length === 0) throw new ValidationError("easyModeCheck: production records are empty");
2267
+ const threshold = opts.passThreshold ?? .5;
2268
+ const tolerance = opts.inflationTolerance ?? .1;
2269
+ const passRate = (records, side) => {
2270
+ let passes = 0;
2271
+ for (const r of records) {
2272
+ const score = finiteOrNull(observedSplitScore(r, "holdout")) ?? finiteOrNull(observedSplitScore(r, "search"));
2273
+ if (score === null) throw new ValidationError(`easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`);
2274
+ if (score >= threshold) passes++;
2275
+ }
2276
+ return passes / records.length;
2277
+ };
2278
+ const simPassRate = passRate(simulated, "simulated");
2279
+ const prodPassRate = passRate(production, "production");
2280
+ const gap = simPassRate - prodPassRate;
2281
+ return {
2282
+ simPassRate,
2283
+ prodPassRate,
2284
+ gap,
2285
+ inflated: gap > tolerance
2286
+ };
2287
+ }
2288
+ //#endregion
2289
+ //#region src/rl/tournament.ts
2290
+ /**
2291
+ * Bradley-Terry MLE via Hunter's MM algorithm.
2292
+ *
2293
+ * Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
2294
+ * where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
2295
+ *
2296
+ * Returns log-strengths normalized so the smallest is 0 (any constant
2297
+ * offset is unobservable in BT — only differences are identified).
2298
+ */
1880
2299
  function fitBradleyTerry(outcomes, opts = {}) {
1881
- const tol = opts.tolerance ?? 1e-6;
1882
- const maxIter = opts.maxIterations ?? 256;
1883
- const smoothing = opts.smoothing ?? 0.1;
1884
- const candidates = /* @__PURE__ */ new Set();
1885
- for (const o of outcomes) {
1886
- candidates.add(o.winner);
1887
- candidates.add(o.loser);
1888
- }
1889
- const ids = [...candidates].sort();
1890
- const idx = new Map(ids.map((id, i) => [id, i]));
1891
- const n = ids.length;
1892
- if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
1893
- if (n === 1) {
1894
- return {
1895
- ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
1896
- iterations: 0,
1897
- finalDelta: 0,
1898
- converged: true
1899
- };
1900
- }
1901
- const W = Array.from({ length: n }, () => new Array(n).fill(0));
1902
- const N = Array.from({ length: n }, () => new Array(n).fill(0));
1903
- for (const o of outcomes) {
1904
- const i = idx.get(o.winner);
1905
- const j = idx.get(o.loser);
1906
- const w = o.weight ?? 1;
1907
- if (o.draw) {
1908
- W[i][j] += 0.5 * w;
1909
- W[j][i] += 0.5 * w;
1910
- } else {
1911
- W[i][j] += w;
1912
- }
1913
- N[i][j] += w;
1914
- N[j][i] += w;
1915
- }
1916
- const winsTotal = new Array(n).fill(0);
1917
- for (let i = 0; i < n; i++) {
1918
- for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
1919
- winsTotal[i] += smoothing;
1920
- }
1921
- const compsTotal = new Array(n).fill(0);
1922
- for (let i = 0; i < n; i++) {
1923
- for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
1924
- }
1925
- let theta = new Array(n).fill(1);
1926
- let iter = 0;
1927
- let delta = Infinity;
1928
- for (; iter < maxIter; iter++) {
1929
- const newTheta = new Array(n);
1930
- for (let i = 0; i < n; i++) {
1931
- let denom = 0;
1932
- for (let j = 0; j < n; j++) {
1933
- if (j === i) continue;
1934
- if (N[i][j] === 0) continue;
1935
- denom += N[i][j] / (theta[i] + theta[j]);
1936
- }
1937
- newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
1938
- }
1939
- let logSum = 0;
1940
- for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
1941
- const norm = Math.exp(logSum / n);
1942
- for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
1943
- delta = 0;
1944
- for (let i = 0; i < n; i++) {
1945
- const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
1946
- if (d > delta) delta = d;
1947
- }
1948
- theta = newTheta;
1949
- if (delta < tol) break;
1950
- }
1951
- const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
1952
- const ratings = ids.map((id, i) => ({
1953
- candidateId: id,
1954
- strength: theta[i],
1955
- logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
1956
- n: compsTotal[i],
1957
- wins: winsTotal[i] - smoothing
1958
- }));
1959
- return {
1960
- ratings: ratings.sort((a, b) => b.strength - a.strength),
1961
- iterations: iter,
1962
- finalDelta: delta,
1963
- converged: delta < tol
1964
- };
2300
+ const tol = opts.tolerance ?? 1e-6;
2301
+ const maxIter = opts.maxIterations ?? 256;
2302
+ const smoothing = opts.smoothing ?? .1;
2303
+ const candidates = /* @__PURE__ */ new Set();
2304
+ for (const o of outcomes) {
2305
+ candidates.add(o.winner);
2306
+ candidates.add(o.loser);
2307
+ }
2308
+ const ids = [...candidates].sort();
2309
+ const idx = new Map(ids.map((id, i) => [id, i]));
2310
+ const n = ids.length;
2311
+ if (n === 0) return {
2312
+ ratings: [],
2313
+ iterations: 0,
2314
+ finalDelta: 0,
2315
+ converged: true
2316
+ };
2317
+ if (n === 1) return {
2318
+ ratings: [{
2319
+ candidateId: ids[0],
2320
+ strength: 1,
2321
+ logStrength: 0,
2322
+ n: 0,
2323
+ wins: 0
2324
+ }],
2325
+ iterations: 0,
2326
+ finalDelta: 0,
2327
+ converged: true
2328
+ };
2329
+ const W = Array.from({ length: n }, () => new Array(n).fill(0));
2330
+ const N = Array.from({ length: n }, () => new Array(n).fill(0));
2331
+ for (const o of outcomes) {
2332
+ const i = idx.get(o.winner);
2333
+ const j = idx.get(o.loser);
2334
+ const w = o.weight ?? 1;
2335
+ if (o.draw) {
2336
+ W[i][j] += .5 * w;
2337
+ W[j][i] += .5 * w;
2338
+ } else W[i][j] += w;
2339
+ N[i][j] += w;
2340
+ N[j][i] += w;
2341
+ }
2342
+ const winsTotal = new Array(n).fill(0);
2343
+ for (let i = 0; i < n; i++) {
2344
+ for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
2345
+ winsTotal[i] += smoothing;
2346
+ }
2347
+ const compsTotal = new Array(n).fill(0);
2348
+ for (let i = 0; i < n; i++) for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
2349
+ let theta = new Array(n).fill(1);
2350
+ let iter = 0;
2351
+ let delta = Infinity;
2352
+ for (; iter < maxIter; iter++) {
2353
+ const newTheta = new Array(n);
2354
+ for (let i = 0; i < n; i++) {
2355
+ let denom = 0;
2356
+ for (let j = 0; j < n; j++) {
2357
+ if (j === i) continue;
2358
+ if (N[i][j] === 0) continue;
2359
+ denom += N[i][j] / (theta[i] + theta[j]);
2360
+ }
2361
+ newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
2362
+ }
2363
+ let logSum = 0;
2364
+ for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
2365
+ const norm = Math.exp(logSum / n);
2366
+ for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
2367
+ delta = 0;
2368
+ for (let i = 0; i < n; i++) {
2369
+ const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
2370
+ if (d > delta) delta = d;
2371
+ }
2372
+ theta = newTheta;
2373
+ if (delta < tol) break;
2374
+ }
2375
+ const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
2376
+ return {
2377
+ ratings: ids.map((id, i) => ({
2378
+ candidateId: id,
2379
+ strength: theta[i],
2380
+ logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
2381
+ n: compsTotal[i],
2382
+ wins: winsTotal[i] - smoothing
2383
+ })).sort((a, b) => b.strength - a.strength),
2384
+ iterations: iter,
2385
+ finalDelta: delta,
2386
+ converged: delta < tol
2387
+ };
1965
2388
  }
1966
2389
  function applyEloUpdate(ratings, outcome, opts = {}) {
1967
- const defaultRating = opts.defaultRating ?? 1500;
1968
- const k = opts.kFactor ?? 32;
1969
- const rW = ratings.get(outcome.winner) ?? defaultRating;
1970
- const rL = ratings.get(outcome.loser) ?? defaultRating;
1971
- const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
1972
- const scoreW = outcome.draw ? 0.5 : 1;
1973
- const scoreL = outcome.draw ? 0.5 : 0;
1974
- const w = outcome.weight ?? 1;
1975
- const winnerDelta = k * w * (scoreW - expectedW);
1976
- const loserDelta = k * w * (scoreL - (1 - expectedW));
1977
- ratings.set(outcome.winner, rW + winnerDelta);
1978
- ratings.set(outcome.loser, rL + loserDelta);
1979
- return { winnerDelta, loserDelta };
2390
+ const defaultRating = opts.defaultRating ?? 1500;
2391
+ const k = opts.kFactor ?? 32;
2392
+ const rW = ratings.get(outcome.winner) ?? defaultRating;
2393
+ const rL = ratings.get(outcome.loser) ?? defaultRating;
2394
+ const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
2395
+ const scoreW = outcome.draw ? .5 : 1;
2396
+ const scoreL = outcome.draw ? .5 : 0;
2397
+ const w = outcome.weight ?? 1;
2398
+ const winnerDelta = k * w * (scoreW - expectedW);
2399
+ const loserDelta = k * w * (scoreL - (1 - expectedW));
2400
+ ratings.set(outcome.winner, rW + winnerDelta);
2401
+ ratings.set(outcome.loser, rL + loserDelta);
2402
+ return {
2403
+ winnerDelta,
2404
+ loserDelta
2405
+ };
1980
2406
  }
1981
2407
  function buildPairwiseFromCampaign(input) {
1982
- const drawMargin = input.drawMargin ?? 0;
1983
- const byKey = /* @__PURE__ */ new Map();
1984
- for (const r of input.runs) {
1985
- const arr = byKey.get(r.matchKey) ?? [];
1986
- arr.push({ candidateId: r.candidateId, score: r.score });
1987
- byKey.set(r.matchKey, arr);
1988
- }
1989
- const outcomes = [];
1990
- for (const arr of byKey.values()) {
1991
- for (let i = 0; i < arr.length; i++) {
1992
- for (let j = i + 1; j < arr.length; j++) {
1993
- const a = arr[i];
1994
- const b = arr[j];
1995
- if (a.candidateId === b.candidateId) continue;
1996
- const margin = Math.abs(a.score - b.score);
1997
- if (margin <= drawMargin) {
1998
- outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
1999
- } else {
2000
- const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
2001
- outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
2002
- }
2003
- }
2004
- }
2005
- }
2006
- return outcomes;
2007
- }
2008
- export {
2009
- ABSENT_CATEGORY,
2010
- DEFAULT_MIN_N_PER_FEATURE,
2011
- DEFAULT_QUANTILE_BUCKETS,
2012
- DPO_CONTEXT_REQUIREMENT,
2013
- FileSystemOutcomeStore,
2014
- InMemoryOutcomeStore,
2015
- PRM_CONTEXT_REQUIREMENT,
2016
- PredictiveValidityResearcher,
2017
- REPRESENTATIVE_MIN_FIDELITY,
2018
- STEP_REWARD_CONTEXT_REQUIREMENT,
2019
- appendToCorpus,
2020
- applyEloUpdate,
2021
- assertPrmTrainableLine,
2022
- bestOfN,
2023
- bucketLabel,
2024
- buildDatasetFromCorpus,
2025
- buildPairwiseFromCampaign,
2026
- buildRlDataset,
2027
- campaignToRunRecords,
2028
- compareAdaptationCurves,
2029
- datasheetToMarkdown,
2030
- defaultBehaviorFeatures,
2031
- detectRewardHacking,
2032
- doublyRobust,
2033
- easyModeCheck,
2034
- extractPreferences,
2035
- extractStepRewards,
2036
- extractVerifiableReward,
2037
- extractVerifiableRewardsFromRecords,
2038
- filterDeterministicallyRewarded,
2039
- firstPassK,
2040
- fitBradleyTerry,
2041
- injectIrrelevantClause,
2042
- inverseProbabilityWeighting,
2043
- jsDivergence,
2044
- observationsFromRunRecords,
2045
- offPolicyEstimateAll,
2046
- paretoFrontier,
2047
- prmTrainingPairs,
2048
- quantileEdges,
2049
- readCorpus,
2050
- renameVariables,
2051
- runAdaptationCurve,
2052
- runComputeCurve,
2053
- runContaminationProbe,
2054
- runEvalCampaign,
2055
- runRLCampaign,
2056
- runwiseStepRewardSummary,
2057
- selfConsistency,
2058
- selfNormalizedImportanceWeighting,
2059
- shuffleOrder,
2060
- simFidelityReport,
2061
- stepRewardsToJsonl,
2062
- thompsonCurriculum,
2063
- toAnthropicFormat,
2064
- toDpoJsonl,
2065
- toDpoRows,
2066
- toGrpoJsonl,
2067
- toGrpoRows,
2068
- toPrmJsonl,
2069
- toPrmRows,
2070
- toSftJsonl,
2071
- toSftRows,
2072
- toTRLFormat,
2073
- validateDatasetFormats,
2074
- varianceBasedCurriculum,
2075
- verificationReportToRunRecord
2076
- };
2408
+ const drawMargin = input.drawMargin ?? 0;
2409
+ const byKey = /* @__PURE__ */ new Map();
2410
+ for (const r of input.runs) {
2411
+ const arr = byKey.get(r.matchKey) ?? [];
2412
+ arr.push({
2413
+ candidateId: r.candidateId,
2414
+ score: r.score
2415
+ });
2416
+ byKey.set(r.matchKey, arr);
2417
+ }
2418
+ const outcomes = [];
2419
+ for (const arr of byKey.values()) for (let i = 0; i < arr.length; i++) for (let j = i + 1; j < arr.length; j++) {
2420
+ const a = arr[i];
2421
+ const b = arr[j];
2422
+ if (a.candidateId === b.candidateId) continue;
2423
+ const margin = Math.abs(a.score - b.score);
2424
+ if (margin <= drawMargin) outcomes.push({
2425
+ winner: a.candidateId,
2426
+ loser: b.candidateId,
2427
+ draw: true,
2428
+ weight: 1
2429
+ });
2430
+ else {
2431
+ const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
2432
+ outcomes.push({
2433
+ winner: winner.candidateId,
2434
+ loser: loser.candidateId,
2435
+ weight: margin
2436
+ });
2437
+ }
2438
+ }
2439
+ return outcomes;
2440
+ }
2441
+ //#endregion
2442
+ export { ABSENT_CATEGORY, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, DPO_CONTEXT_REQUIREMENT, FileSystemOutcomeStore, InMemoryOutcomeStore, PRM_CONTEXT_REQUIREMENT, PredictiveValidityResearcher, REPRESENTATIVE_MIN_FIDELITY, STEP_REWARD_CONTEXT_REQUIREMENT, appendToCorpus, applyEloUpdate, assertPrmTrainableLine, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };
2443
+
2077
2444
  //# sourceMappingURL=rl.js.map