@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
package/dist/rl.d.ts CHANGED
@@ -1,525 +1,47 @@
1
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
2
- type AgentProfileJsonObject = {
3
- [key: string]: AgentProfileJson;
4
- };
5
- type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | AgentProfileJsonObject;
6
- type AgentProfileDimensionValue = string | number | boolean | null;
7
- interface AgentProfileSource {
8
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
9
- kind: string;
10
- /** sha256 over the canonical source profile object. */
11
- hash: string;
12
- }
13
- interface AgentProfileSourceInput {
14
- kind: string;
15
- /** Precomputed sha256 for callers that already sign their profile artifact. */
16
- hash?: string;
17
- /** Full canonical runtime profile; hashed and then discarded from the cell. */
18
- profile?: AgentProfileJson;
19
- }
20
- interface AgentProfileHarness {
21
- id: string;
22
- version?: string;
23
- hash?: string;
24
- }
25
- interface AgentProfileCellInput {
26
- profileId: string;
27
- sourceProfile: AgentProfileSourceInput;
28
- harness?: AgentProfileHarness;
29
- model?: string;
30
- promptHash?: string;
31
- dimensions?: Record<string, AgentProfileDimensionValue>;
32
- }
33
- interface AgentProfileCell {
34
- schemaVersion: AgentProfileCellSchemaVersion;
35
- cellId: string;
36
- profileId: string;
37
- sourceProfile: AgentProfileSource;
38
- harness?: AgentProfileHarness;
39
- model?: string;
40
- promptHash?: string;
41
- dimensions?: Record<string, AgentProfileDimensionValue>;
42
- }
43
-
44
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
45
- interface BudgetSpec {
46
- tokens?: number;
47
- wallMs?: number;
48
- calls?: number;
49
- usd?: number;
50
- }
51
- interface RunOutcome$1 {
52
- score?: number;
53
- pass?: boolean;
54
- failureClass?: FailureClass;
55
- notes?: string;
56
- }
57
- /**
58
- * Layer — optional classification in a nested build workflow.
59
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
60
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
61
- * `app-runtime`: a run of the generated agent against a domain scenario.
62
- * `meta`: any meta-eval (judge replay, correlation analysis).
63
- */
64
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
65
- interface Run {
66
- runId: string;
67
- /**
68
- * Stable identifier of the scenario being executed.
69
- *
70
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
71
- * input WITHOUT this field, substituting a sensible default
72
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
73
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
74
- * keeps the persisted shape unambiguous for downstream filters + aggregations
75
- * while removing the boilerplate of inventing placeholder ids at the call site.
76
- */
77
- scenarioId: string;
78
- variantId?: string;
79
- datasetVersion?: string;
80
- /** Git SHA of agent code at run time. */
81
- codeSha?: string;
82
- /** Hash of the prompt template + any system prompt. */
83
- promptSha?: string;
84
- /** Model id + date + system-prompt hash, concatenated. */
85
- modelFingerprint?: string;
86
- seed?: number;
87
- /** Arbitrary environment markers (shell, docker version, tz). */
88
- envFingerprint?: Record<string, string>;
89
- /** Version of the redaction rules applied to this run. */
90
- redactionVersion?: string;
91
- /** Parent run in a nested build workflow. A builder run's children are
92
- * app-build runs; those children are app-runtime runs. */
93
- parentRunId?: string;
94
- /** Stable project identifier — groups runs across chats + sessions. */
95
- projectId?: string;
96
- /** Chat/conversation identifier within a project. */
97
- chatId?: string;
98
- /** Layer classification — hint for aggregation; not enforced. */
99
- layer?: RunLayer;
100
- startedAt: number;
101
- endedAt?: number;
102
- status: RunStatus;
103
- outcome?: RunOutcome$1;
104
- budget?: BudgetSpec;
105
- /** Free-form labels for downstream grouping. */
106
- tags?: Record<string, string>;
107
- }
108
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
109
- type SpanStatus = 'ok' | 'error';
110
- interface SpanBase {
111
- spanId: string;
112
- parentSpanId?: string;
113
- runId: string;
114
- kind: SpanKind;
115
- name: string;
116
- startedAt: number;
117
- endedAt?: number;
118
- status?: SpanStatus;
119
- error?: string;
120
- /** Anything not covered by typed fields. Kept deliberately free-form. */
121
- attributes?: Record<string, unknown>;
122
- }
123
- interface Message {
124
- role: 'system' | 'user' | 'assistant' | 'tool';
125
- content: string;
126
- tokens?: number;
127
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
128
- images?: Array<{
129
- artifactId?: string;
130
- url?: string;
131
- mime?: string;
132
- }>;
133
- }
134
- interface LlmSpan extends SpanBase {
135
- kind: 'llm';
136
- model: string;
137
- messages: Message[];
138
- output?: string;
139
- inputTokens?: number;
140
- /** All generated tokens, including the reasoning subset when present. */
141
- outputTokens?: number;
142
- cachedTokens?: number;
143
- cacheWriteTokens?: number;
144
- /** Reasoning-token subset of `outputTokens`. */
145
- reasoningTokens?: number;
146
- costUsd?: number;
147
- finishReason?: string;
148
- }
149
- interface ToolSpan extends SpanBase {
150
- kind: 'tool';
151
- toolName: string;
152
- args: unknown;
153
- /** False when the source observed the call but did not capture its arguments. */
154
- argsCaptured?: boolean;
155
- result?: unknown;
156
- latencyMs?: number;
157
- }
158
- interface RetrievalSpan extends SpanBase {
159
- kind: 'retrieval';
160
- query: string;
161
- hits: Array<{
162
- docId: string;
163
- score: number;
164
- content?: string;
165
- }>;
166
- }
167
- interface JudgeSpan extends SpanBase {
168
- kind: 'judge';
169
- judgeId: string;
170
- /** Span this judgment applies to. */
171
- targetSpanId: string;
172
- dimension: string;
173
- /** Numeric score (free-range; interpretation up to the judge). */
174
- score: number;
175
- rationale?: string;
176
- evidence?: string;
177
- }
178
- interface SandboxSpan extends SpanBase {
179
- kind: 'sandbox';
180
- image?: string;
181
- command?: string;
182
- exitCode?: number;
183
- testsTotal?: number;
184
- testsPassed?: number;
185
- stdoutHash?: string;
186
- stderrHash?: string;
187
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
188
- wallMs?: number;
189
- }
190
- interface GenericSpan extends SpanBase {
191
- kind: 'agent' | 'custom';
192
- }
193
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
194
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
195
- interface TraceEvent {
196
- eventId: string;
197
- runId: string;
198
- spanId?: string;
199
- kind: EventKind;
200
- timestamp: number;
201
- payload: Record<string, unknown>;
202
- }
203
- interface BudgetLedgerEntry {
204
- runId: string;
205
- dimension: keyof BudgetSpec;
206
- limit: number;
207
- consumed: number;
208
- remaining: number;
209
- timestamp: number;
210
- breached: boolean;
211
- /** Span that triggered this entry, if any. */
212
- spanId?: string;
213
- }
214
- interface Artifact {
215
- artifactId: string;
216
- runId: string;
217
- spanId?: string;
218
- contentType: string;
219
- sizeBytes: number;
220
- /** sha256 in hex. */
221
- hash: string;
222
- /** External storage URL (R2, S3, filesystem path). */
223
- storageUrl?: string;
224
- /** Inline content for small blobs — keep under ~64KB. */
225
- inlineContent?: string;
226
- }
227
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
228
-
229
- /**
230
- * Paper-grade RunRecord schema + runtime validator.
231
- *
232
- * Every run that participates in a promotion gate, paper table, or
233
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
234
- * fields are exactly those the paper "Two Loops, Three Roles" requires
235
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
236
- * holdout split tag. A task score is optional because execution-only records
237
- * must preserve missing labels instead of converting errors into zero quality.
238
- *
239
- * This is intentionally NOT a replacement for the rich `Run` /
240
- * `ProposeReviewReport` / `ScenarioResult` types already in the
241
- * package. Those are runtime structures with full provenance. A
242
- * `RunRecord` is the analysis-time projection — the JSON-friendly
243
- * row you'd put in a parquet file or paste into a notebook.
244
- *
245
- * Validate at the boundary:
246
- *
247
- * const rec = validateRunRecord(rawJson) // throws on missing
248
- * const ok = isRunRecord(rawJson) // boolean check
249
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
250
- *
251
- * The validator runs in pure TS — zod is intentionally NOT a
252
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
253
- */
254
-
255
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
256
- * combined train+test pool that the optimizer is allowed to read. */
257
- type RunSplitTag = 'search' | 'dev' | 'holdout';
258
- /**
259
- * Explicit execution-lifecycle result for a run.
260
- *
261
- * This is separate from task quality (`outcome`) and failure classification.
262
- * Producers set it only from root-run or process evidence.
263
- */
264
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
265
- interface RunTokenUsage {
266
- input: number;
267
- /** All generated tokens charged as output, including reasoning tokens. */
268
- output: number;
269
- /** Reasoning-token subset of `output`, when the provider reports it. */
270
- reasoning?: number;
271
- /** Prompt tokens served from a provider cache. */
272
- cached?: number;
273
- /** Prompt tokens written into a provider cache. */
274
- cacheWrite?: number;
275
- }
276
- /**
277
- * How a run's USD amount was obtained.
278
- */
279
- type RunCostProvenance = {
280
- kind: 'observed';
281
- usd: number;
282
- } | {
283
- kind: 'estimated';
284
- usd: number;
285
- } | {
286
- kind: 'uncaptured';
287
- usd: null;
288
- };
289
- interface RunJudgeMetadata {
290
- model: string;
291
- promptVersion: string;
292
- /** [0,1] confidence the judge declared. Constant judge confidence
293
- * across many runs is a fallback signal (see `canary.ts`). */
294
- confidence: number;
295
- /** True if the judge degraded to a fallback path (rules-only,
296
- * prior-call cache, etc.). The canary uses this to alert. */
297
- fallback: boolean;
298
- }
299
- /**
300
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
301
- * judges over a multi-dimensional rubric.
302
- *
303
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
304
- * composite the gate uses. The full breakdown belongs here so consumers
305
- * can answer "which judge disagreed?", "which dimension dragged the
306
- * composite down?", and "did half the panel fail?" without re-running.
307
- *
308
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
309
- * `composite` are convenience projections — derivable but precomputed so
310
- * downstream IRR primitives (`interRaterReliability`,
311
- * `corpusInterRaterAgreement`) and reporters don't pay the same
312
- * aggregation twice.
313
- *
314
- * Fail-loud discipline: judges that errored out land in `failedJudges`
315
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
316
- * run); the explicit list makes a partial-failure recorded as such.
317
- */
318
- interface JudgeScoresRecord {
319
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
320
- perJudge: Record<string, Record<string, number>>;
321
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
322
- perDimMean: Record<string, number>;
323
- /** Composite mean across successful judges. Mirrors the task score only
324
- * when `failedJudges` is empty. */
325
- composite: number;
326
- /** Judges that errored or returned an unparseable verdict. Recorded
327
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
328
- * not inferred from missing keys in `perJudge`. */
329
- failedJudges?: string[];
330
- /** Free-form notes the judges emitted (joined across judges or
331
- * first-judge only — consumer's choice). */
332
- notes?: string;
333
- }
334
- interface RunOutcome {
335
- /** Score on the search/optimization split. Optional for holdout-only and
336
- * execution-only records. */
337
- searchScore?: number;
338
- /** Score on the held-out split. Optional for search-only and execution-only
339
- * records. When both scores are absent, the run is explicitly unlabeled. */
340
- holdoutScore?: number;
341
- /** Bag of any other metric the run produced — judge dimensions,
342
- * pass/fail counters, latency stats, etc. Numeric only — keeps
343
- * reporters honest. */
344
- raw: Record<string, number>;
345
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
346
- * judgements populate this; substrate primitives like
347
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
348
- * these records as input. Optional — single-judge or scalar-only
349
- * runs leave it unset. */
350
- judgeScores?: JudgeScoresRecord;
351
- /** Authenticity / realness verdict — did the run build the REAL thing on the
352
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
353
- * with an authenticity config populate it. Carried in the corpus so the
354
- * flywheel / off-policy learning can optimize for real completion, not gamed
355
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
356
- * must not count as a real success regardless of `score`. */
357
- realness?: {
358
- score: number;
359
- gated: boolean;
360
- reason?: string;
361
- };
362
- }
363
- /**
364
- * Mandatory paper-grade fields for a single evaluation run. Optional
365
- * fields are extension points; mandatory fields throw if missing.
366
- *
367
- * Hash discipline:
368
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
369
- * model (after any steering bundle merge).
370
- * - `configHash` is the sha256 of the effective run config (model,
371
- * temperature, tools, judges, splits). The pair (promptHash,
372
- * configHash) uniquely identifies an experiment cell.
373
- *
374
- * Model snapshot discipline:
375
- * - `model` MUST encode a snapshot version. Bare aliases like
376
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
377
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
378
- */
379
- interface RunRecord {
380
- /** UUID for the run. */
381
- runId: string;
382
- /** Logical experiment grouping (a treatment vs a baseline within
383
- * the same sweep should share `experimentId`). */
384
- experimentId: string;
385
- /** Stable identifier for the candidate (variant) being run. The
386
- * promotion gate compares two `candidateId`s on matched items. */
387
- candidateId: string;
388
- /** RNG seed for the run. Always recorded — silent re-seeding is
389
- * the most common cause of non-reproducible numbers. */
390
- seed: number;
391
- /** Model identifier WITH snapshot version. */
392
- model: string;
393
- /** sha256 of the effective prompt (post-steering). */
394
- promptHash: string;
395
- /** sha256 of the effective config. */
396
- configHash: string;
397
- /** Git SHA the harness was run from. */
398
- commitSha: string;
399
- /** End-to-end wall-clock duration in milliseconds. */
400
- wallMs: number;
401
- /** Time spent queued before execution started, if known. */
402
- queueMs?: number;
403
- /** Total USD cost, or null when the producer could not capture one. */
404
- costUsd: number | null;
405
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
406
- costProvenance: RunCostProvenance;
407
- /** Token usage breakdown. */
408
- tokenUsage: RunTokenUsage;
409
- /** Root-run or process terminal result. Never inferred from a child span. */
410
- terminalOutcome: RunTerminalOutcome;
411
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
412
- * or incomplete terminal result; never populated from a child span. */
413
- terminalFailureReason?: string;
414
- /** Judge-side metadata, if a judge was used. */
415
- judgeMetadata?: RunJudgeMetadata;
416
- /** Per-split scores + raw bag. */
417
- outcome: RunOutcome;
418
- /** Canonical task-failure class drawn from the shared
419
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
420
- * evidence. Execution errors belong in
421
- * `outcome.raw.execution_error_count`. */
422
- failureClass?: FailureClass;
423
- /** Free-form task-failure detail scoped under a non-success
424
- * `failureClass`. It is invalid without that class. */
425
- failureMode?: string;
426
- /** Which split this run was drawn from. */
427
- splitTag: RunSplitTag;
428
- /**
429
- * Stable scenario identifier the run observed or was scored against.
430
- * Comparison primitives match this identity rather than input order.
431
- */
432
- scenarioId: string;
433
- /**
434
- * Canonical identity for the agent profile cell that produced this row:
435
- * profile artifact hash plus optional harness/model/prompt/reporting
436
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
437
- * longitudinal reports by the complete source profile, not by a loose
438
- * candidate label or opaque config hash.
439
- */
440
- agentProfile?: AgentProfileCell;
441
- }
442
- /**
443
- * Canonical task-result classification.
444
- *
445
- * A producer may omit classification, record explicit success, or attach
446
- * domain-specific detail to a non-success class. Detail can never stand alone.
447
- * Execution errors belong in `outcome.raw.execution_error_count`.
448
- */
449
- type RunTaskFailure = {
450
- failureClass?: undefined;
451
- failureMode?: undefined;
452
- } | {
453
- failureClass: 'success';
454
- failureMode?: undefined;
455
- } | {
456
- failureClass: Exclude<FailureClass, 'success'>;
457
- failureMode?: string;
458
- };
459
-
460
- /**
461
- * Adaptive curriculum / active scenario selection.
462
- *
463
- * Fixed scenario sets waste sample budget on cells the policy already
464
- * passes (no information left) and cells the policy never passes (no
465
- * gradient available either). Active learning over scenarios fixes this
466
- * by allocating the next sample budget to cells where the policy's
467
- * outcome is *uncertain* — those carry the most decision-relevant signal.
468
- *
469
- * This module ships two complementary strategies:
470
- *
471
- * 1. **Variance-based** — score each (variant, scenario) cell by the
472
- * empirical variance of past observations. Allocate next-round budget
473
- * proportional to variance. Standard active-learning-by-uncertainty
474
- * heuristic; works well when the policy is non-deterministic and
475
- * cells differ in observation noise.
476
- *
477
- * 2. **Bandit-based (Thompson sampling)** — model each (variant,
478
- * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
479
- * cells whose posterior mean is closest to the per-scenario decision
480
- * threshold. The right primitive when scenarios are
481
- * "pass/fail" rather than continuous, and when promotion gates fire
482
- * at a known threshold (e.g., 0.5).
483
- *
484
- * The output is a *next-round budget allocation* — a list of (variant,
485
- * scenario, count) triples. The consumer's matrix runner consumes the
486
- * allocation, runs those cells, feeds the new observations back. Loop.
487
- *
488
- * Out of scope (deliberate): scenario *generation* — that's the
489
- * adversarial primitive's job. This module allocates over an existing
490
- * scenario pool.
491
- */
492
-
1
+ import { s as VerificationReport } from "./multi-layer-verifier-BHY1gWAc.js";
2
+ import { a as RunRecord, s as RunSplitTag } from "./run-record-CnZu_gjl.js";
3
+ import { _ as Span } from "./schema-BtVldJ3T.js";
4
+ import { s as TraceStore } from "./store-CT9YIIve.js";
5
+ import { i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "./outcome-store-BYHIuO0e.js";
6
+ import { r as RubricPredictiveValidityReport } from "./rubric-predictive-validity-Ku_clp_1.js";
7
+ import { a as doublyRobust, c as selfNormalizedImportanceWeighting, i as OffPolicyTrajectory, n as OffPolicyEstimate, o as inverseProbabilityWeighting, r as OffPolicyOptions, s as offPolicyEstimateAll, t as OffPolicyContributionCounts } from "./off-policy-mskQw8Mb.js";
8
+ import { a as CampaignResult } from "./types-k9tZGKUg.js";
9
+ import { a as detectRewardHacking, c as VerifiableRewardSource, d as filterDeterministicallyRewarded, i as RewardHackingSignal, l as extractVerifiableReward, n as RewardHackingFinding, o as VerifiableReward, r as RewardHackingReport, s as VerifiableRewardExtractionOptions, t as DetectRewardHackingInput, u as extractVerifiableRewardsFromRecords } from "./reward-hacking-eAnOsynk.js";
10
+ import { t as AdversarialMutation } from "./adversarial-smnADNFS.js";
11
+ import { b as RolloutSplit, o as MintedRolloutLine } from "./schema-Cef2cFmb.js";
12
+ import { _ as EvalCampaignResult, a as FailureMode, c as SteeringChange, g as EvalCampaignOptions, i as ExperimentResult, r as ExperimentPlan, s as Researcher, y as runEvalCampaign } from "./researcher-CwTdwXG1.js";
13
+ import { t as InterimReleaseConfidence } from "./sequential-CYwq6Ff_.js";
14
+ //#region src/rl/active-curriculum.d.ts
493
15
  interface CellObservation {
494
- variantId: string;
495
- scenarioId: string;
496
- /** Observed score in [0, 1]. */
497
- score: number;
498
- /** For Bernoulli arms — derive from the score with a threshold if needed. */
499
- pass?: boolean;
16
+ variantId: string;
17
+ scenarioId: string;
18
+ /** Observed score in [0, 1]. */
19
+ score: number;
20
+ /** For Bernoulli arms — derive from the score with a threshold if needed. */
21
+ pass?: boolean;
500
22
  }
501
23
  interface CurriculumAllocation {
502
- variantId: string;
503
- scenarioId: string;
504
- /** How many additional reps to run on this cell. */
505
- count: number;
506
- /** Strategy-specific reason for the allocation. */
507
- reason: string;
24
+ variantId: string;
25
+ scenarioId: string;
26
+ /** How many additional reps to run on this cell. */
27
+ count: number;
28
+ /** Strategy-specific reason for the allocation. */
29
+ reason: string;
508
30
  }
509
31
  interface VarianceCurriculumOptions {
510
- /** Total reps to allocate across all cells. */
511
- budget: number;
512
- /**
513
- * Smoothing prior on variance — keeps the allocator from concentrating
514
- * on a cell with one observation just because its 1-sample variance is
515
- * 0. Default 0.05.
516
- */
517
- variancePrior?: number;
518
- /**
519
- * Minimum reps per cell — even when the variance estimate is low, give
520
- * every cell at least this many. Default 1.
521
- */
522
- floorPerCell?: number;
32
+ /** Total reps to allocate across all cells. */
33
+ budget: number;
34
+ /**
35
+ * Smoothing prior on variance — keeps the allocator from concentrating
36
+ * on a cell with one observation just because its 1-sample variance is
37
+ * 0. Default 0.05.
38
+ */
39
+ variancePrior?: number;
40
+ /**
41
+ * Minimum reps per cell — even when the variance estimate is low, give
42
+ * every cell at least this many. Default 1.
43
+ */
44
+ floorPerCell?: number;
523
45
  }
524
46
  /**
525
47
  * Variance-proportional allocation. For each cell, estimate variance from
@@ -529,22 +51,22 @@ interface VarianceCurriculumOptions {
529
51
  * under-sampled cells."
530
52
  */
531
53
  declare function varianceBasedCurriculum(observations: CellObservation[], candidateCells: Array<{
532
- variantId: string;
533
- scenarioId: string;
54
+ variantId: string;
55
+ scenarioId: string;
534
56
  }>, opts: VarianceCurriculumOptions): CurriculumAllocation[];
535
57
  interface ThompsonCurriculumOptions {
536
- budget: number;
537
- /**
538
- * The per-scenario decision threshold. Cells whose posterior mean is
539
- * closest to this get the most budget — that's where the next observation
540
- * has the highest information value for the gate decision. Default 0.5.
541
- */
542
- decisionThreshold?: number;
543
- /** Beta prior parameters. Default α=β=1 (uniform). */
544
- priorAlpha?: number;
545
- priorBeta?: number;
546
- /** Seed the Thompson sampler. Default unset (Math.random). */
547
- seed?: number;
58
+ budget: number;
59
+ /**
60
+ * The per-scenario decision threshold. Cells whose posterior mean is
61
+ * closest to this get the most budget — that's where the next observation
62
+ * has the highest information value for the gate decision. Default 0.5.
63
+ */
64
+ decisionThreshold?: number;
65
+ /** Beta prior parameters. Default α=β=1 (uniform). */
66
+ priorAlpha?: number;
67
+ priorBeta?: number;
68
+ /** Seed the Thompson sampler. Default unset (Math.random). */
69
+ seed?: number;
548
70
  }
549
71
  /**
550
72
  * Thompson-sampling-style allocation for pass/fail cells. For each cell:
@@ -558,15 +80,16 @@ interface ThompsonCurriculumOptions {
558
80
  * threshold and you want to sharpen the posterior near the boundary.
559
81
  */
560
82
  declare function thompsonCurriculum(observations: CellObservation[], candidateCells: Array<{
561
- variantId: string;
562
- scenarioId: string;
83
+ variantId: string;
84
+ scenarioId: string;
563
85
  }>, opts: ThompsonCurriculumOptions): CurriculumAllocation[];
564
86
  /** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
565
87
  declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
566
- passThreshold?: number;
567
- useHoldout?: boolean;
88
+ passThreshold?: number;
89
+ useHoldout?: boolean;
568
90
  }): CellObservation[];
569
-
91
+ //#endregion
92
+ //#region src/rl/adaptation-eval.d.ts
570
93
  /**
571
94
  * Sample-efficient adaptation evaluation.
572
95
  *
@@ -595,73 +118,73 @@ declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
595
118
  * - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
596
119
  */
597
120
  interface AdaptationRunner<S> {
598
- /**
599
- * Runs the policy on `scenario` with `k` demonstrations. Returns a
600
- * scalar score in [0, 1]. The runner is responsible for any caching;
601
- * the harness calls it once per (scenario, k, rep) cell.
602
- */
603
- run(args: {
604
- scenario: S;
605
- k: number;
606
- rep: number;
607
- }): Promise<number>;
121
+ /**
122
+ * Runs the policy on `scenario` with `k` demonstrations. Returns a
123
+ * scalar score in [0, 1]. The runner is responsible for any caching;
124
+ * the harness calls it once per (scenario, k, rep) cell.
125
+ */
126
+ run(args: {
127
+ scenario: S;
128
+ k: number;
129
+ rep: number;
130
+ }): Promise<number>;
608
131
  }
609
132
  interface RunAdaptationCurveOptions<S> {
610
- scenarios: S[];
611
- /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
612
- ks?: number[];
613
- /** Reps per (scenario, k) cell. Default 3. */
614
- reps?: number;
615
- runner: AdaptationRunner<S>;
616
- /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
617
- passThreshold?: number;
133
+ scenarios: S[];
134
+ /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
135
+ ks?: number[];
136
+ /** Reps per (scenario, k) cell. Default 3. */
137
+ reps?: number;
138
+ runner: AdaptationRunner<S>;
139
+ /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
140
+ passThreshold?: number;
618
141
  }
619
142
  interface AdaptationPoint {
620
- k: number;
143
+ k: number;
144
+ meanScore: number;
145
+ passRate: number;
146
+ std: number;
147
+ n: number;
148
+ /** Per-scenario means at this k. */
149
+ perScenario: Array<{
150
+ scenarioId: string;
621
151
  meanScore: number;
622
- passRate: number;
623
- std: number;
624
- n: number;
625
- /** Per-scenario means at this k. */
626
- perScenario: Array<{
627
- scenarioId: string;
628
- meanScore: number;
629
- passes: number;
630
- total: number;
631
- }>;
152
+ passes: number;
153
+ total: number;
154
+ }>;
632
155
  }
633
156
  interface AdaptationCurve {
634
- points: AdaptationPoint[];
635
- /**
636
- * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
637
- * tested reaches it.
638
- */
639
- firstPassK: number | null;
640
- /**
641
- * Area under the (k, meanScore) curve, normalized by max-k. A
642
- * single-number summary of "how well does this policy adapt from
643
- * cold-start to fully-conditioned." Higher = better adapter.
644
- */
645
- adaptationArea: number;
157
+ points: AdaptationPoint[];
158
+ /**
159
+ * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
160
+ * tested reaches it.
161
+ */
162
+ firstPassK: number | null;
163
+ /**
164
+ * Area under the (k, meanScore) curve, normalized by max-k. A
165
+ * single-number summary of "how well does this policy adapt from
166
+ * cold-start to fully-conditioned." Higher = better adapter.
167
+ */
168
+ adaptationArea: number;
646
169
  }
647
170
  declare function runAdaptationCurve<S extends {
648
- scenarioId?: string;
171
+ scenarioId?: string;
649
172
  }>(opts: RunAdaptationCurveOptions<S>): Promise<AdaptationCurve>;
650
173
  interface CompareCurvesResult {
651
- perK: Array<{
652
- k: number;
653
- deltaMean: number;
654
- aLow: number;
655
- aHigh: number;
656
- bLow: number;
657
- bHigh: number;
658
- }>;
659
- areaDelta: number;
660
- firstPassKDelta: number | null;
661
- /** Verdict: 'a_better' | 'b_better' | 'similar'. */
662
- verdict: 'a_better' | 'b_better' | 'similar';
663
- /** Rationale, ready to render. */
664
- rationale: string;
174
+ perK: Array<{
175
+ k: number;
176
+ deltaMean: number;
177
+ aLow: number;
178
+ aHigh: number;
179
+ bLow: number;
180
+ bHigh: number;
181
+ }>;
182
+ areaDelta: number;
183
+ firstPassKDelta: number | null;
184
+ /** Verdict: 'a_better' | 'b_better' | 'similar'. */
185
+ verdict: 'a_better' | 'b_better' | 'similar';
186
+ /** Rationale, ready to render. */
187
+ rationale: string;
665
188
  }
666
189
  /**
667
190
  * Paired comparison of two adaptation curves. Per-k deltas with 95%
@@ -669,31 +192,14 @@ interface CompareCurvesResult {
669
192
  * — the bootstrap unit is the scenario, not the rep).
670
193
  */
671
194
  declare function compareAdaptationCurves(a: AdaptationCurve, b: AdaptationCurve, opts?: {
672
- confidence?: number;
673
- bootstrapResamples?: number;
674
- seed?: number;
195
+ confidence?: number;
196
+ bootstrapResamples?: number;
197
+ seed?: number;
675
198
  }): CompareCurvesResult;
676
199
  /** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
677
200
  declare function firstPassK(curve: AdaptationCurve, threshold?: number): number | null;
678
-
679
- /**
680
- * Adversarial mutation contract.
681
- *
682
- * `AdversarialMutation<S>` is the scenario-mutation strategy the fuzz harness
683
- * (`fuzzAgent`, src/fuzz) drives: paraphrase, edge-case substitution, or
684
- * compositional combination of a scenario the policy currently passes, looking
685
- * for the tail inputs that break it. The harness supplies the loop; consumers
686
- * supply the mutations and the failure detector.
687
- */
688
- interface AdversarialMutation<S> {
689
- id: string;
690
- /**
691
- * Mutate one scenario. Return null to skip; return one or more new
692
- * scenarios. The harness deduplicates by `mutateScenarioId(scenario)`.
693
- */
694
- mutate(parent: S, rng: () => number): Promise<S[]> | S[];
695
- }
696
-
201
+ //#endregion
202
+ //#region src/rl/compute-curves.d.ts
697
203
  /**
698
204
  * Test-time compute scaling curves.
699
205
  *
@@ -725,82 +231,82 @@ interface AdversarialMutation<S> {
725
231
  * is on whatever axis they pick.
726
232
  */
727
233
  interface ComputeCurveBudget {
728
- /** Identifier — for the report. Common: '1x', '4x', '16x'. */
729
- id: string;
730
- /** Numeric value on the chosen axis (tokens, calls, USD, ms — caller picks). */
731
- cost: number;
732
- /** Free-form metadata (the caller can carry per-budget config). */
733
- meta?: Record<string, unknown>;
234
+ /** Identifier — for the report. Common: '1x', '4x', '16x'. */
235
+ id: string;
236
+ /** Numeric value on the chosen axis (tokens, calls, USD, ms — caller picks). */
237
+ cost: number;
238
+ /** Free-form metadata (the caller can carry per-budget config). */
239
+ meta?: Record<string, unknown>;
734
240
  }
735
241
  interface ComputeCurvePoint {
736
- budgetId: string;
737
- cost: number;
738
- score: number;
739
- /** Number of underlying samples used at this budget. */
740
- samples: number;
741
- /** Optional spread / variance information. */
742
- std?: number;
743
- /** Any extra metrics the runner returned. */
744
- metrics?: Record<string, number>;
242
+ budgetId: string;
243
+ cost: number;
244
+ score: number;
245
+ /** Number of underlying samples used at this budget. */
246
+ samples: number;
247
+ /** Optional spread / variance information. */
248
+ std?: number;
249
+ /** Any extra metrics the runner returned. */
250
+ metrics?: Record<string, number>;
745
251
  }
746
252
  interface ComputeCurve {
747
- candidateId: string;
748
- points: ComputeCurvePoint[];
749
- /** Rough exponent fit: score ≈ a + b * log(cost). Useful for "how steep is the curve?" */
750
- logSlope: number | null;
751
- /** Best (highest-score) point on the curve. */
752
- best: ComputeCurvePoint;
253
+ candidateId: string;
254
+ points: ComputeCurvePoint[];
255
+ /** Rough exponent fit: score ≈ a + b * log(cost). Useful for "how steep is the curve?" */
256
+ logSlope: number | null;
257
+ /** Best (highest-score) point on the curve. */
258
+ best: ComputeCurvePoint;
753
259
  }
754
260
  interface RunComputeCurveOptions {
755
- candidateId: string;
756
- budgets: ComputeCurveBudget[];
757
- /**
758
- * Run the candidate at one budget. Returns the realized score plus
759
- * optional spread + extra metrics.
760
- */
761
- runAtBudget: (budget: ComputeCurveBudget) => Promise<{
762
- score: number;
763
- samples: number;
764
- std?: number;
765
- metrics?: Record<string, number>;
766
- }>;
261
+ candidateId: string;
262
+ budgets: ComputeCurveBudget[];
263
+ /**
264
+ * Run the candidate at one budget. Returns the realized score plus
265
+ * optional spread + extra metrics.
266
+ */
267
+ runAtBudget: (budget: ComputeCurveBudget) => Promise<{
268
+ score: number;
269
+ samples: number;
270
+ std?: number;
271
+ metrics?: Record<string, number>;
272
+ }>;
767
273
  }
768
274
  declare function runComputeCurve(opts: RunComputeCurveOptions): Promise<ComputeCurve>;
769
275
  interface ComputeBestOfNOptions<O> {
770
- /** Number of independent samples to draw. */
771
- n: number;
772
- /** Sampler — produces one rollout. */
773
- sample: (sampleIdx: number) => Promise<O>;
774
- /** Score one rollout. */
775
- scoreFn: (rollout: O) => Promise<number> | number;
276
+ /** Number of independent samples to draw. */
277
+ n: number;
278
+ /** Sampler — produces one rollout. */
279
+ sample: (sampleIdx: number) => Promise<O>;
280
+ /** Score one rollout. */
281
+ scoreFn: (rollout: O) => Promise<number> | number;
776
282
  }
777
283
  interface ComputeBestOfNResult<O> {
778
- best: O;
779
- bestScore: number;
780
- scores: number[];
781
- meanScore: number;
782
- /** Index of the best rollout, for diagnostics. */
783
- bestIndex: number;
284
+ best: O;
285
+ bestScore: number;
286
+ scores: number[];
287
+ meanScore: number;
288
+ /** Index of the best rollout, for diagnostics. */
289
+ bestIndex: number;
784
290
  }
785
291
  /** The simplest test-time scaling primitive. */
786
292
  declare function bestOfN<O>(opts: ComputeBestOfNOptions<O>): Promise<ComputeBestOfNResult<O>>;
787
293
  interface SelfConsistencyOptions<O> {
788
- n: number;
789
- sample: (sampleIdx: number) => Promise<O>;
790
- /** Extract the canonical answer key (string) from a rollout. */
791
- answerKey: (rollout: O) => string;
294
+ n: number;
295
+ sample: (sampleIdx: number) => Promise<O>;
296
+ /** Extract the canonical answer key (string) from a rollout. */
297
+ answerKey: (rollout: O) => string;
792
298
  }
793
299
  interface SelfConsistencyResult<O> {
794
- /** Modal answer (the majority vote). */
795
- answer: string;
796
- /** Fraction of samples voting for the modal answer in [0, 1]. */
797
- agreement: number;
798
- /** Histogram of all answers. */
799
- histogram: Record<string, number>;
800
- /** A representative rollout that voted for the modal answer. */
801
- representative: O;
802
- /** All rollouts. */
803
- rollouts: O[];
300
+ /** Modal answer (the majority vote). */
301
+ answer: string;
302
+ /** Fraction of samples voting for the modal answer in [0, 1]. */
303
+ agreement: number;
304
+ /** Histogram of all answers. */
305
+ histogram: Record<string, number>;
306
+ /** A representative rollout that voted for the modal answer. */
307
+ representative: O;
308
+ /** All rollouts. */
309
+ rollouts: O[];
804
310
  }
805
311
  /**
806
312
  * Self-consistency / majority-vote test-time scaling. For tasks with a
@@ -814,13 +320,14 @@ declare function selfConsistency<O>(opts: SelfConsistencyOptions<O>): Promise<Se
814
320
  * by cost.
815
321
  */
816
322
  interface ParetoPointInput {
817
- candidateId: string;
818
- budgetId: string;
819
- cost: number;
820
- score: number;
323
+ candidateId: string;
324
+ budgetId: string;
325
+ cost: number;
326
+ score: number;
821
327
  }
822
328
  declare function paretoFrontier(points: ParetoPointInput[]): ParetoPointInput[];
823
-
329
+ //#endregion
330
+ //#region src/rl/contamination.d.ts
824
331
  /**
825
332
  * Contamination probe — held-out perturbation tests.
826
333
  *
@@ -852,64 +359,64 @@ declare function paretoFrontier(points: ParetoPointInput[]): ParetoPointInput[];
852
359
  */
853
360
  type ScenarioPerturbationKind = 'rename_variables' | 'shuffle_order' | 'paraphrase' | 'inject_irrelevant_clause' | 'custom';
854
361
  interface ScenarioPerturbation<S> {
855
- kind: ScenarioPerturbationKind;
856
- /** Apply to one scenario, return its perturbed sibling. */
857
- apply: (scenario: S) => Promise<S> | S;
858
- /** Optional id — for the report. */
859
- id?: string;
362
+ kind: ScenarioPerturbationKind;
363
+ /** Apply to one scenario, return its perturbed sibling. */
364
+ apply: (scenario: S) => Promise<S> | S;
365
+ /** Optional id — for the report. */
366
+ id?: string;
860
367
  }
861
368
  interface ContaminationProbeInput<S> {
862
- /** Identity of every scenario. The probe's `runFingerprint` keys on these. */
863
- scenarioId: (s: S) => string;
864
- /** Original scenarios. */
865
- originals: S[];
866
- /**
867
- * Either pre-computed perturbations (one per original, same order) OR a
868
- * `perturbation` strategy that synthesizes them on the fly.
869
- */
870
- perturbed?: S[];
871
- perturbation?: ScenarioPerturbation<S>;
872
- /**
873
- * Run the policy/agent against one scenario and return a scalar score
874
- * in [0, 1]. The probe doesn't care what the policy is — that's the
875
- * caller's contract.
876
- */
877
- scoreFn: (s: S) => Promise<number>;
369
+ /** Identity of every scenario. The probe's `runFingerprint` keys on these. */
370
+ scenarioId: (s: S) => string;
371
+ /** Original scenarios. */
372
+ originals: S[];
373
+ /**
374
+ * Either pre-computed perturbations (one per original, same order) OR a
375
+ * `perturbation` strategy that synthesizes them on the fly.
376
+ */
377
+ perturbed?: S[];
378
+ perturbation?: ScenarioPerturbation<S>;
379
+ /**
380
+ * Run the policy/agent against one scenario and return a scalar score
381
+ * in [0, 1]. The probe doesn't care what the policy is — that's the
382
+ * caller's contract.
383
+ */
384
+ scoreFn: (s: S) => Promise<number>;
878
385
  }
879
386
  interface ContaminationProbeOptions {
880
- /** Drop scores below this from the probe; treats partial failures separately. Default 0. */
881
- scoreFloor?: number;
882
- /**
883
- * BH-FDR threshold for declaring contamination on each per-scenario
884
- * delta. Default 0.05.
885
- */
886
- fdr?: number;
887
- /**
888
- * Minimum median per-scenario drop to flag global contamination. Default
889
- * 0.05 (5 percentage points). Smaller drops may be noise.
890
- */
891
- minMedianDrop?: number;
387
+ /** Drop scores below this from the probe; treats partial failures separately. Default 0. */
388
+ scoreFloor?: number;
389
+ /**
390
+ * BH-FDR threshold for declaring contamination on each per-scenario
391
+ * delta. Default 0.05.
392
+ */
393
+ fdr?: number;
394
+ /**
395
+ * Minimum median per-scenario drop to flag global contamination. Default
396
+ * 0.05 (5 percentage points). Smaller drops may be noise.
397
+ */
398
+ minMedianDrop?: number;
892
399
  }
893
400
  interface ContaminationProbeReport {
894
- perScenario: Array<{
895
- scenarioId: string;
896
- originalScore: number;
897
- perturbedScore: number;
898
- delta: number;
899
- /** Per-scenario q-value (single-test BH for a single scenario). Mainly for display. */
900
- qValue: number;
901
- }>;
902
- /** Wilcoxon paired-test on the deltas. */
903
- pairedTest: {
904
- w: number;
905
- p: number;
906
- };
907
- medianDelta: number;
908
- meanDelta: number;
909
- contaminationSuspected: boolean;
910
- reason: string;
911
- /** Number of scenarios processed. */
912
- n: number;
401
+ perScenario: Array<{
402
+ scenarioId: string;
403
+ originalScore: number;
404
+ perturbedScore: number;
405
+ delta: number;
406
+ /** Per-scenario q-value (single-test BH for a single scenario). Mainly for display. */
407
+ qValue: number;
408
+ }>;
409
+ /** Wilcoxon paired-test on the deltas. */
410
+ pairedTest: {
411
+ w: number;
412
+ p: number;
413
+ };
414
+ medianDelta: number;
415
+ meanDelta: number;
416
+ contaminationSuspected: boolean;
417
+ reason: string;
418
+ /** Number of scenarios processed. */
419
+ n: number;
913
420
  }
914
421
  declare function runContaminationProbe<S>(input: ContaminationProbeInput<S>, opts?: ContaminationProbeOptions): Promise<ContaminationProbeReport>;
915
422
  /**
@@ -919,7 +426,7 @@ declare function runContaminationProbe<S>(input: ContaminationProbeInput<S>, opt
919
426
  * (e.g. SWE-Bench-style coding tasks).
920
427
  */
921
428
  declare function renameVariables<S extends {
922
- prompt: string;
429
+ prompt: string;
923
430
  }>(identifiers: string[], rename?: (name: string, idx: number) => string): ScenarioPerturbation<S>;
924
431
  /**
925
432
  * Order-shuffle perturbation. Reshuffles a list-shaped section of the
@@ -927,7 +434,7 @@ declare function renameVariables<S extends {
927
434
  * on the option labels, not order). Caller provides the section extractor.
928
435
  */
929
436
  declare function shuffleOrder<S extends {
930
- prompt: string;
437
+ prompt: string;
931
438
  }>(shuffleSection: (prompt: string, rng: () => number) => string, seed: number): ScenarioPerturbation<S>;
932
439
  /**
933
440
  * Inject-irrelevant-clause perturbation. Adds a benign sentence that
@@ -935,286 +442,236 @@ declare function shuffleOrder<S extends {
935
442
  * the input string."
936
443
  */
937
444
  declare function injectIrrelevantClause<S extends {
938
- prompt: string;
445
+ prompt: string;
939
446
  }>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
940
-
941
- /**
942
- * Preference dataset extraction — bridge from `RunRecord[]` to RL training.
943
- *
944
- * Production RLHF / DPO / KTO / SimPO pipelines need preference triples:
945
- * `(prompt, chosen, rejected)`. The campaign artifact already contains the
946
- * ingredients every (variantId, scenarioId, seed) cell is a candidate
947
- * that ran the same prompt against the same scenario, scored by the same
948
- * judge but turning that into a clean preference dataset requires
949
- * deciding *what counts as a preference*.
950
- *
951
- * This module ships three preference-extraction strategies with explicit
952
- * tradeoffs, plus a unified output type compatible with HuggingFace TRL,
953
- * Anthropic finetuning JSONL, and OpenAI fine-tuning APIs. The strategies
954
- * are deliberately not auto-magical picking the wrong one corrupts the
955
- * gradient.
956
- *
957
- * Strategies:
958
- *
959
- * 1. **`paired-by-scenario-and-seed`** exact-match comparisons. For
960
- * each scenario × seed pair, compare every (variantA, variantB) on
961
- * that exact (scenario, seed). Matches scenarios so the comparison
962
- * isolates variant effects. Highest signal-to-noise; smallest
963
- * dataset (only matched pairs count).
964
- *
965
- * 2. **`paired-by-scenario`** looser matching. For each scenario,
966
- * compare every (variantA, variantB) where both have ≥ 1 run on the
967
- * same scenario. Aggregates across seeds to compute mean scores per
968
- * (variant, scenario), then forms preferences from the means. More
969
- * data, lower per-pair signal.
970
- *
971
- * 3. **`top-vs-bottom`** coarsest. Within each scenario, the highest-
972
- * scoring run is `chosen`, the lowest is `rejected`. Smallest dataset
973
- * per scenario but biggest score gap per pair. Useful for early
974
- * bootstrapping when you have few variants.
975
- *
976
- * Resolve `PreferenceTriple` text with `toDpoRows` from `./exporters`.
977
- */
978
-
447
+ //#endregion
448
+ //#region src/rl/rollout-input.d.ts
449
+ /**
450
+ * The minted lines behind a LINE-LESS training artifact.
451
+ *
452
+ * `PreferenceTriple`, `PrmTrainingTriple` and `StepReward` all carry a bare
453
+ * reward number plus run ids, and nothing that says whether those runs faked
454
+ * their success. An exporter over them therefore has no way, from its input
455
+ * alone, to learn that its chosen side is a run the gate flagged — it will
456
+ * happily emit the gaming trajectory as the preferred one. Supplying the lines
457
+ * is what gives it eyes.
458
+ */
459
+ interface RolloutLineContext {
460
+ /**
461
+ * Minted lines for every INVOCATION the artifacts reference.
462
+ *
463
+ * Not "one line per run": `tangle.rollout.v1` models many invocations per
464
+ * `run_id` — that is what `rollout_id` and `parent_rollout_id` are for, and
465
+ * `supervisorRunRolloutLines` emits a supervisor node plus one per worker, all
466
+ * sharing a single `run_id`. A reference is resolved against `rollout_id`
467
+ * first and falls back to `run_id` only when that run has exactly one
468
+ * invocation; see `resolveInvocation`.
469
+ */
470
+ lines: MintedRolloutLine[];
471
+ }
472
+ /** How one exporter names itself and its context type in the failure messages. */
473
+ interface LineContextRequirement {
474
+ /** Exporter label, e.g. `'DPO export'`. */
475
+ exporter: string;
476
+ /** The context type the caller must pass, e.g. `'DpoLineContext'`. */
477
+ contextType: string;
478
+ /** Why this exporter cannot see the gate without lines. One sentence. */
479
+ because: string;
480
+ }
481
+ //#endregion
482
+ //#region src/rl/preferences.d.ts
979
483
  type PreferenceStrategy = 'paired-by-scenario-and-seed' | 'paired-by-scenario' | 'top-vs-bottom';
980
484
  interface PreferenceTriple {
981
- /** The scenario (input) the variants were run against. */
982
- scenarioId: string;
983
- /** RunRecord ids on each side, for traceability. */
984
- chosenRunId: string;
985
- rejectedRunId: string;
986
- /** Variant ids — load-bearing for the RL update. */
987
- chosenVariantId: string;
988
- rejectedVariantId: string;
989
- /** The score gap between chosen and rejected. Larger = stronger signal. */
990
- marginScore: number;
991
- /**
992
- * Optional `(chosen_score, rejected_score)` pair for soft-margin DPO
993
- * variants. Omitted for `top-vs-bottom` runs that don't carry meaningful
994
- * scalar gaps.
995
- */
996
- scores?: {
997
- chosen: number;
998
- rejected: number;
999
- };
1000
- /** Tie-breaker — when multiple seeds match this scenario, the one used. */
1001
- seed?: number;
1002
- /**
1003
- * Free-form metadata propagated from the run records e.g. original
1004
- * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.
1005
- */
1006
- meta: {
1007
- chosenPromptHash: string;
1008
- rejectedPromptHash: string;
1009
- chosenConfigHash: string;
1010
- rejectedConfigHash: string;
1011
- chosenModel: string;
1012
- rejectedModel: string;
1013
- };
1014
- }
1015
- interface ExtractPreferencesOptions extends TrainingRunSelectionOptions {
1016
- strategy?: PreferenceStrategy;
1017
- /**
1018
- * Minimum score gap required to admit a pair. Pairs below this are
1019
- * dropped — they're noise, not signal. Default 0.05 (5% of [0,1]).
1020
- */
1021
- minMargin?: number;
1022
- /**
1023
- * Optional split tag filter. Without one, only search is included.
1024
- * Holdout requires `allowHeldOutTrainingData: true`; dev is evaluation-only.
1025
- */
1026
- splitTag?: RunRecord['splitTag'];
1027
- /**
1028
- * Optional reward extractor that overrides `outcome.holdoutScore` /
1029
- * `outcome.searchScore`. Use to drive preferences off a verifiable
1030
- * reward instead of the headline score.
1031
- */
1032
- rewardOf?: (run: RunRecord) => number | null;
485
+ /** The scenario (input) the variants were run against. */
486
+ scenarioId: string;
487
+ /** RunRecord ids on each side, for traceability. */
488
+ chosenRunId: string;
489
+ rejectedRunId: string;
490
+ /** Variant ids — load-bearing for the RL update. */
491
+ chosenVariantId: string;
492
+ rejectedVariantId: string;
493
+ /** The score gap between chosen and rejected. Larger = stronger signal. */
494
+ marginScore: number;
495
+ /**
496
+ * Optional `(chosen_score, rejected_score)` pair for soft-margin DPO
497
+ * variants. Omitted for `top-vs-bottom` runs that don't carry meaningful
498
+ * scalar gaps.
499
+ */
500
+ scores?: {
501
+ chosen: number;
502
+ rejected: number;
503
+ };
504
+ /** Tie-breaker — when multiple seeds match this scenario, the one used. */
505
+ seed?: number;
506
+ /**
507
+ * Free-form metadata propagated from the rollout lines, such as original
508
+ * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.
509
+ */
510
+ meta: {
511
+ chosenPromptHash: string;
512
+ rejectedPromptHash: string;
513
+ chosenConfigHash: string;
514
+ rejectedConfigHash: string;
515
+ chosenModel: string;
516
+ rejectedModel: string;
517
+ };
518
+ }
519
+ interface ExtractPreferencesOptions {
520
+ strategy?: PreferenceStrategy;
521
+ /**
522
+ * Minimum score gap required to admit a pair. Pairs below this are
523
+ * dropped — they're noise, not signal. Default 0.05 (5% of [0,1]).
524
+ */
525
+ minMargin?: number;
526
+ /**
527
+ * Optional split filter. Without one, only search is included.
528
+ * Holdout requires `allowHeldOutTrainingData: true`; dev and canary are
529
+ * evaluation-only.
530
+ */
531
+ split?: RolloutSplit;
532
+ /** Named opt-in required before held-out lines may be paired. */
533
+ allowHeldOutTrainingData?: boolean;
1033
534
  }
1034
535
  interface PreferenceExtractionReport {
1035
- pairs: PreferenceTriple[];
1036
- /** Number of (scenario, seed) cells inspected. */
1037
- cellsInspected: number;
1038
- /** Number of pairs filtered by `minMargin`. */
1039
- pairsBelowMargin: number;
1040
- /** Number of cells with only one variant (no comparison possible). */
1041
- cellsSingleton: number;
1042
- /** Strategy used. */
1043
- strategy: PreferenceStrategy;
1044
- }
1045
- /**
1046
- * Convert `RunRecord[]` to preference triples for RL training.
536
+ pairs: PreferenceTriple[];
537
+ /** Number of (scenario, seed) cells inspected. */
538
+ cellsInspected: number;
539
+ /** Number of pairs filtered by `minMargin`. */
540
+ pairsBelowMargin: number;
541
+ /** Number of cells with only one variant (no comparison possible). */
542
+ cellsSingleton: number;
543
+ /** Strategy used. */
544
+ strategy: PreferenceStrategy;
545
+ /**
546
+ * Lines dropped before pairing because they carry no `candidate_id`. A
547
+ * preference is a statement about two candidates, so a line that names none
548
+ * cannot be paired.
549
+ */
550
+ linesWithoutCandidateId: number;
551
+ }
552
+ /**
553
+ * Convert rollout lines to preference triples for RL training.
1047
554
  *
1048
555
  * Returns a structured report so callers can see how much data was
1049
556
  * dropped and why (low-margin pairs, singleton cells). For production
1050
557
  * pipelines, you usually want to:
1051
558
  *
1052
559
  * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
1053
- * 2. Call this with `strategy: 'paired-by-scenario-and-seed'` and a
1054
- * verifiable-reward extractor as `rewardOf`
1055
- * 3. Pass `report.pairs` to `toDpoRows` with prompt/completion resolvers
1056
- */
1057
- declare function extractPreferences(runs: RunRecord[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
560
+ * 2. Mint the runs with `mintRolloutRows` and call this with
561
+ * `strategy: 'paired-by-scenario-and-seed'`
562
+ * 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with
563
+ * prompt/completion resolvers and pipe to your DPO trainer
564
+ *
565
+ * The gate is what makes a preference dataset safe: ordered on an ungated
566
+ * score, a gamed run with an inflated number becomes the `chosen` side and DPO
567
+ * is trained to prefer the gaming trajectory over its honest sibling. A gated
568
+ * line arrives here already scored 0, so it sinks to `rejected`.
569
+ */
570
+ declare function extractPreferences(lines: MintedRolloutLine[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
571
+ /**
572
+ * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
573
+ * where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes
574
+ * would optimize the policy toward emitting hex digests. Neither the prompt
575
+ * nor the completions live on the triple (it carries only run ids and hashes),
576
+ * so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`
577
+ * takes, keyed by run id, and this function resolves real text.
578
+ *
579
+ * The chosen and rejected sides of a valid pair share one prompt; resolving
580
+ * both and comparing catches lookup bugs (a stale map keyed by the wrong id)
581
+ * before they ship a row whose prompt does not match its rejected completion.
582
+ *
583
+ * `context` is REQUIRED: this is the third exporter over the identical
584
+ * line-less input class, and the round that hardened `toPrmRows` while leaving
585
+ * `toDpoRows` open is why every one of them now takes the same argument and
586
+ * runs the same admission rule.
587
+ */
588
+ declare function toTRLFormat(triples: PreferenceTriple[], lookups: DpoLookups, context: RolloutLineContext): Promise<Array<{
589
+ prompt: string;
590
+ chosen: string;
591
+ rejected: string;
592
+ }>>;
1058
593
  /**
1059
594
  * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
1060
595
  * shape. Same caveat as TRL: prompt + outputs are content the caller has
1061
596
  * to map back from the run record / raw event log.
1062
- */
1063
- declare function toAnthropicFormat(triples: PreferenceTriple[]): Array<{
1064
- scenarioId: string;
1065
- chosenRunId: string;
1066
- rejectedRunId: string;
1067
- margin: number;
1068
- }>;
1069
-
1070
- interface RunFilter {
1071
- scenarioId?: string;
1072
- variantId?: string;
1073
- status?: RunStatus;
1074
- since?: number;
1075
- until?: number;
1076
- tag?: {
1077
- key: string;
1078
- value: string;
1079
- };
1080
- parentRunId?: string;
1081
- projectId?: string;
1082
- chatId?: string;
1083
- layer?: RunLayer;
1084
- }
1085
- interface SpanFilter {
1086
- runId?: string;
1087
- parentSpanId?: string;
1088
- kind?: SpanKind;
1089
- name?: string;
1090
- toolName?: string;
1091
- judgeId?: string;
1092
- since?: number;
1093
- until?: number;
1094
- }
1095
- interface EventFilter {
1096
- runId?: string;
1097
- spanId?: string;
1098
- kind?: EventKind;
1099
- since?: number;
1100
- until?: number;
1101
- }
1102
- interface TraceStore {
1103
- appendRun(run: Run): Promise<void>;
1104
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
1105
- appendSpan(span: Span): Promise<void>;
1106
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
1107
- appendEvent(event: TraceEvent): Promise<void>;
1108
- appendArtifact(artifact: Artifact): Promise<void>;
1109
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
1110
- getRun(runId: string): Promise<Run | undefined>;
1111
- listRuns(filter?: RunFilter): Promise<Run[]>;
1112
- spans(filter?: SpanFilter): Promise<Span[]>;
1113
- events(filter?: EventFilter): Promise<TraceEvent[]>;
1114
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
1115
- artifacts(runId: string): Promise<Artifact[]>;
1116
- }
1117
-
1118
- /**
1119
- * Process reward extraction — step-level credit assignment from trace spans.
1120
- *
1121
- * RL on long-horizon agents needs *step-level* rewards, not run-level
1122
- * ones. The classic credit-assignment problem (Sutton & Barto) requires
1123
- * knowing which sub-decisions in a trajectory contributed to the
1124
- * outcome. Modern systems (DeepSeek-R1, OpenAI o-series, Lightman et al.
1125
- * "Let's Verify Step by Step" 2023) train *process reward models* (PRMs)
1126
- * that score every step, then do RL with the PRM as the reward signal.
1127
- *
1128
- * This module extracts `StepReward[]` from trace spans — one per
1129
- * meaningful step — and ships:
1130
- *
1131
- * 1. `extractStepRewards(store, runId, opts)` — span → step-reward
1132
- * conversion using configurable per-span scorers (LLM judge over the
1133
- * span output, deterministic checkers, or a learned PRM).
1134
- * 2. `runwiseStepRewardSummary(stepRewards)` — aggregate the per-step
1135
- * signal into a credit-assignment-aware run-level score.
1136
- * 3. `prmTrainingPairs(stepRewards, options)` — produce the
1137
- * `(prefix, suffix_chosen, suffix_rejected)` triples that PRM
1138
- * training pipelines consume.
1139
597
  *
1140
- * What we ship: the *extraction* and *aggregation* infrastructure plus
1141
- * the data shape PRM training expects. We do NOT ship the actual PRM
1142
- * training (gradient descent over a transformer is out of scope for a
1143
- * TS package). The interface is the contract; downstream consumers wire
1144
- * their preferred trainer.
1145
- *
1146
- * Caveat the panel will land: this is descriptive credit assignment
1147
- * (which steps correlate with outcome), not causal credit assignment
1148
- * (which steps caused outcome). For causal claims you need
1149
- * counterfactual rollouts or a learned dynamics model. Future work; the
1150
- * descriptive version is what production PRM training actually uses.
598
+ * `context` is REQUIRED see `toTRLFormat`. The emitted `margin` is a number
599
+ * derived from the two runs' rewards, so this row is training signal even
600
+ * though it ships no completion text.
1151
601
  */
1152
-
602
+ declare function toAnthropicFormat(triples: PreferenceTriple[], context: RolloutLineContext): Array<{
603
+ scenarioId: string;
604
+ chosenRunId: string;
605
+ rejectedRunId: string;
606
+ margin: number;
607
+ }>;
608
+ //#endregion
609
+ //#region src/rl/process-reward.d.ts
1153
610
  interface StepReward {
1154
- /** Trace span this reward attaches to. */
1155
- spanId: string;
1156
- runId: string;
1157
- /** Index in the trajectory (0-based, in started-at order). */
1158
- stepIndex: number;
1159
- /** Span kind (typically 'tool', 'llm', 'judge'). */
1160
- kind: Span['kind'];
1161
- /** Span name — for the consumer's downstream filtering. */
1162
- name: string;
1163
- /** Step-level reward in [0, 1]. */
1164
- reward: number;
1165
- /**
1166
- * Determinism class. Mirrors the verifiable-reward distinction:
1167
- * deterministic = test/compile/schema check; probabilistic = LLM judge.
1168
- */
1169
- determinism: 'deterministic' | 'probabilistic';
1170
- /** Optional rationale / evidence — the trainer typically discards. */
1171
- rationale?: string;
1172
- /** Optional weight — how much this step contributes to credit assignment. */
1173
- weight?: number;
611
+ /** Trace span this reward attaches to. */
612
+ spanId: string;
613
+ runId: string;
614
+ /** Index in the trajectory (0-based, in started-at order). */
615
+ stepIndex: number;
616
+ /** Span kind (typically 'tool', 'llm', 'judge'). */
617
+ kind: Span['kind'];
618
+ /** Span name — for the consumer's downstream filtering. */
619
+ name: string;
620
+ /** Step-level reward in [0, 1]. */
621
+ reward: number;
622
+ /**
623
+ * Determinism class. Mirrors the verifiable-reward distinction:
624
+ * deterministic = test/compile/schema check; probabilistic = LLM judge.
625
+ */
626
+ determinism: 'deterministic' | 'probabilistic';
627
+ /** Optional rationale / evidence — the trainer typically discards. */
628
+ rationale?: string;
629
+ /** Optional weight — how much this step contributes to credit assignment. */
630
+ weight?: number;
1174
631
  }
1175
632
  interface StepScorer {
1176
- /** Span kinds this scorer applies to. */
1177
- appliesTo: Span['kind'][];
1178
- /** Returns null to skip the span; returns a `StepReward` shape (without index/runId/spanId, which are filled in). */
1179
- score(span: Span): Promise<Omit<StepReward, 'spanId' | 'runId' | 'stepIndex'>> | null | undefined;
633
+ /** Span kinds this scorer applies to. */
634
+ appliesTo: Span['kind'][];
635
+ /** Returns null to skip the span; returns a `StepReward` shape (without index/runId/spanId, which are filled in). */
636
+ score(span: Span): Promise<Omit<StepReward, 'spanId' | 'runId' | 'stepIndex'>> | null | undefined;
1180
637
  }
1181
638
  interface ExtractStepRewardsOptions {
1182
- /**
1183
- * Ordered list of scorers. Each span runs through scorers in order;
1184
- * the first non-null result wins. If no scorer applies, the span is
1185
- * skipped (not all spans are training-worthy).
1186
- */
1187
- scorers: StepScorer[];
1188
- /** Optional filter — return null to drop the span entirely before scoring. */
1189
- preFilter?: (span: Span) => boolean;
639
+ /**
640
+ * Ordered list of scorers. Each span runs through scorers in order;
641
+ * the first non-null result wins. If no scorer applies, the span is
642
+ * skipped (not all spans are training-worthy).
643
+ */
644
+ scorers: StepScorer[];
645
+ /** Optional filter — return null to drop the span entirely before scoring. */
646
+ preFilter?: (span: Span) => boolean;
1190
647
  }
1191
648
  declare function extractStepRewards(store: TraceStore, runId: string, opts: ExtractStepRewardsOptions): Promise<StepReward[]>;
1192
649
  interface RunwiseStepSummary {
1193
- runId: string;
1194
- totalSteps: number;
1195
- meanReward: number;
1196
- /** Sum-of-rewards (weighted by `weight ?? 1`). Use as the run-level proxy. */
1197
- sumWeightedReward: number;
1198
- /** Fraction of steps where reward < 0.5 — proxy for "where the policy was wrong." */
1199
- failureFraction: number;
1200
- /** Maximum drop in reward between consecutive steps — diagnoses a step where things went sideways. */
1201
- worstStepDelta: number;
1202
- worstStepIndex: number | null;
650
+ runId: string;
651
+ totalSteps: number;
652
+ meanReward: number;
653
+ /** Sum-of-rewards (weighted by `weight ?? 1`). Use as the run-level proxy. */
654
+ sumWeightedReward: number;
655
+ /** Fraction of steps where reward < 0.5 — proxy for "where the policy was wrong." */
656
+ failureFraction: number;
657
+ /** Maximum drop in reward between consecutive steps — diagnoses a step where things went sideways. */
658
+ worstStepDelta: number;
659
+ worstStepIndex: number | null;
1203
660
  }
1204
661
  declare function runwiseStepRewardSummary(stepRewards: StepReward[]): RunwiseStepSummary;
1205
662
  interface PrmTrainingTriple {
1206
- /** Prefix run-id (or composite key) — the trajectory up to step k-1. */
1207
- prefixRunId: string;
1208
- prefixStepIndex: number;
1209
- /** The step that came next on a high-reward trajectory. */
1210
- chosenSpanId: string;
1211
- chosenReward: number;
1212
- /** A step from a divergent low-reward trajectory at the same prefix length. */
1213
- rejectedSpanId: string;
1214
- rejectedReward: number;
1215
- /** The prefix run came from this run; the rejected step came from `rejectedRunId`. */
1216
- rejectedRunId: string;
1217
- marginScore: number;
663
+ /** Prefix run-id (or composite key) — the trajectory up to step k-1. */
664
+ prefixRunId: string;
665
+ prefixStepIndex: number;
666
+ /** The step that came next on a high-reward trajectory. */
667
+ chosenSpanId: string;
668
+ chosenReward: number;
669
+ /** A step from a divergent low-reward trajectory at the same prefix length. */
670
+ rejectedSpanId: string;
671
+ rejectedReward: number;
672
+ /** The prefix run came from this run; the rejected step came from `rejectedRunId`. */
673
+ rejectedRunId: string;
674
+ marginScore: number;
1218
675
  }
1219
676
  /**
1220
677
  * Build PRM training triples. The shape: pair runs that share an early
@@ -1232,289 +689,319 @@ interface PrmTrainingTriple {
1232
689
  * hash; the heuristic is good enough for early-stage scaffolding.
1233
690
  */
1234
691
  declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, opts?: {
1235
- minMargin?: number;
1236
- minPrefixLength?: number;
692
+ minMargin?: number;
693
+ minPrefixLength?: number;
1237
694
  }): PrmTrainingTriple[];
1238
-
1239
- /**
1240
- * Trainer-format exporters.
1241
- *
1242
- * agent-eval produces canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`,
1243
- * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
1244
- * different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
1245
- * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
1246
- * JSONL conventions. Rather than ship N adapters, this module ships the
1247
- * canonical formats most production pipelines accept and ergonomic helpers
1248
- * for the rest.
1249
- *
1250
- * Shapes:
1251
- * - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed
1252
- * by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.
1253
- * - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.
1254
- * Consumed by prime-rl GRPO, verl, OpenRLHF.
1255
- * - **SFT** — `{messages[]}` JSONL with chosen completion as the final
1256
- * assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,
1257
- * Anthropic finetuning.
1258
- * - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.
1259
- * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
1260
- *
1261
- * Why ship this in agent-eval rather than a separate adapter package: the
1262
- * canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`, etc.) are
1263
- * agent-eval's contract; without first-party exporters consumers reverse-
1264
- * engineer the mapping every release. The exporters codify it.
1265
- *
1266
- * The exporters take callbacks for any field that isn't on the canonical
1267
- * artifact (specifically: prompt + completion text, since the package
1268
- * stores only their hashes by design — full text is the consumer's
1269
- * trace store / raw event log).
1270
- */
1271
-
695
+ //#endregion
696
+ //#region src/rl/exporters.d.ts
1272
697
  interface DpoLookups {
1273
- /** Resolve the prompt text for a run (typically from a trace store / raw event sink). */
1274
- promptOf: (runId: string) => string | Promise<string>;
1275
- /** Resolve the assistant completion text for a run. */
1276
- completionOf: (runId: string) => string | Promise<string>;
698
+ /** Resolve the prompt text for a run (typically from a trace store / raw event sink). */
699
+ promptOf: (runId: string) => string | Promise<string>;
700
+ /** Resolve the assistant completion text for a run. */
701
+ completionOf: (runId: string) => string | Promise<string>;
1277
702
  }
1278
703
  interface DpoExportRow {
1279
- prompt: string;
1280
- chosen: string;
1281
- rejected: string;
1282
- /** Carried-through margin. Some KTO / IPO variants use this. */
1283
- margin?: number;
1284
- /** Free-form metadata for downstream filtering / sharding. */
1285
- meta?: Record<string, unknown>;
1286
- }
704
+ prompt: string;
705
+ chosen: string;
706
+ rejected: string;
707
+ /** Carried-through margin. Some KTO / IPO variants use this. */
708
+ margin?: number;
709
+ /** Free-form metadata for downstream filtering / sharding. */
710
+ meta?: Record<string, unknown>;
711
+ }
712
+ /** The minted lines for the runs a `PreferenceTriple` names on each side. */
713
+ type DpoLineContext = RolloutLineContext;
714
+ declare const DPO_CONTEXT_REQUIREMENT: LineContextRequirement;
1287
715
  /**
1288
716
  * Convert preference triples to TRL-compatible DPO rows. The shape
1289
717
  * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
1290
718
  * entry; every major DPO trainer accepts it.
1291
- */
1292
- declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups): Promise<DpoExportRow[]>;
719
+ *
720
+ * `context` is REQUIRED, and for the same reason it is required on the sibling
721
+ * `toPrmRows`: a triple is a line-less artifact. It names two run ids and a
722
+ * margin, and nothing on it says whether either run was flagged as gamed —
723
+ * so a two-argument call applied NO gate at all and emitted the row verbatim,
724
+ * reachable straight through the published bundle builder
725
+ * (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).
726
+ * Triples whose chosen or rejected side is realness-gated are dropped; a triple
727
+ * naming a run with no supplied line is refused. See `admitUngatedByInvocation` for
728
+ * why dropping, not zeroing, is the right disposition for a preference pair.
729
+ */
730
+ declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups, context: DpoLineContext): Promise<DpoExportRow[]>;
1293
731
  /** Serialize DPO rows as JSONL. One line per row. */
1294
732
  declare function toDpoJsonl(rows: DpoExportRow[]): string;
1295
- interface TrainingRunSelectionOptions {
1296
- /** Include held-out evaluation data in training output. Default false. */
1297
- allowHeldOutTrainingData?: boolean;
1298
- /** Require quality to be strictly greater than this value. Default 0. */
1299
- minimumQualityExclusive?: number;
1300
- }
1301
- interface GrpoLookups extends TrainingRunSelectionOptions {
1302
- promptOf: (runId: string) => string | Promise<string>;
1303
- completionOf: (runId: string) => string | Promise<string>;
1304
- /** Optional: derive a custom reward from the run. Defaults to score. */
1305
- rewardOf?: (run: RunRecord) => number | null;
733
+ interface TrainingLineSelectionOptions {
734
+ /** Include held-out evaluation data in training output. Default false. */
735
+ allowHeldOutTrainingData?: boolean;
736
+ /** Require quality to be strictly greater than this value. Default 0. */
737
+ minimumQualityExclusive?: number;
738
+ /**
739
+ * Explicit split selection, replacing the default trainable-split rule.
740
+ * Use this only when producing a deliberately named non-training slice.
741
+ */
742
+ splitFilter?: RolloutSplit[];
743
+ }
744
+ interface GrpoLookups extends Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'> {
745
+ /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */
746
+ promptOf: (runId: string) => string | Promise<string>;
747
+ /** Resolve the assistant completion text for a rollout. */
748
+ completionOf: (runId: string) => string | Promise<string>;
1306
749
  }
1307
750
  interface GrpoExportRow {
1308
- prompt: string;
1309
- completions: string[];
1310
- rewards: number[];
1311
- /** runIds in the same order as `completions[]` for traceability. */
1312
- runIds: string[];
1313
- meta?: Record<string, unknown>;
751
+ prompt: string;
752
+ completions: string[];
753
+ rewards: number[];
754
+ /** runIds in the same order as `completions[]` for traceability. */
755
+ runIds: string[];
756
+ meta?: Record<string, unknown>;
1314
757
  }
1315
758
  /**
1316
- * Convert RunRecord[] grouped by canonical `(scenarioId, promptHash)` identity
1317
- * into GRPO offline rows.
759
+ * Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —
760
+ * one row per scenario, with one completion per rollout on that scenario.
761
+ * A scenario with fewer than two rewarded completions emits no row because a
762
+ * group of one has no relative baseline.
1318
763
  *
1319
764
  * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
1320
765
  * within a group of completions for the same prompt; this is the
1321
- * canonical input format. A scenario containing multiple prompt hashes, or a
1322
- * prompt hash that resolves to different text, is rejected rather than mixed.
766
+ * canonical input format. That relative baseline is exactly why the gate has
767
+ * to hold here: one gamed sibling exporting at full reward shifts the advantage
768
+ * of every honest run beside it.
769
+ *
770
+ * On the line path a realness-gated line stays in its group at reward 0 rather
771
+ * than being dropped. 0 is the honest label for a faked success and is usable
772
+ * signal; removing the line would also move the group's baseline, just in the
773
+ * other direction. (SFT differs — see `toSftRows`.)
1323
774
  */
1324
- declare function toGrpoRows(runs: RunRecord[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
775
+ declare function toGrpoRows(lines: MintedRolloutLine[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
1325
776
  declare function toGrpoJsonl(rows: GrpoExportRow[]): string;
1326
- interface SftLookups extends TrainingRunSelectionOptions {
1327
- promptOf: (runId: string) => string | Promise<string>;
1328
- completionOf: (runId: string) => string | Promise<string>;
1329
- /** Optional system message. Default omits. */
1330
- systemOf?: (run: RunRecord) => string | null | undefined;
1331
- /** Filter return false to skip the run (e.g., low score, failed cases). */
1332
- include?: (run: RunRecord) => boolean;
777
+ interface SftLookups extends TrainingLineSelectionOptions {
778
+ /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */
779
+ promptOf: (runId: string) => string | Promise<string>;
780
+ /** Resolve the assistant completion text for a rollout. */
781
+ completionOf: (runId: string) => string | Promise<string>;
782
+ /** Optional system message. Default omits. */
783
+ systemOf?: (line: MintedRolloutLine) => string | null | undefined;
784
+ /** Extra filter on top of the realness gate (e.g., low score, failed cases). */
785
+ include?: (line: MintedRolloutLine) => boolean;
786
+ /** Include held-out lines under the default split rule. Default false. */
787
+ allowHeldOutTrainingData?: boolean;
1333
788
  }
1334
789
  interface SftExportRow {
1335
- messages: Array<{
1336
- role: 'system' | 'user' | 'assistant';
1337
- content: string;
1338
- }>;
1339
- meta?: Record<string, unknown>;
790
+ messages: Array<{
791
+ role: 'system' | 'user' | 'assistant';
792
+ content: string;
793
+ }>;
794
+ meta?: Record<string, unknown>;
1340
795
  }
1341
796
  /**
1342
- * Convert RunRecord[] into Hugging Face / OpenAI / Anthropic-style
1343
- * conversational SFT rows. By default, only completed, positive-quality
1344
- * search runs are eligible. Pass `include` for additional filtering.
797
+ * Convert rollout lines into Hugging Face / OpenAI / Anthropic-style
798
+ * conversational SFT rows. By default every qualifying line becomes one row;
799
+ * pass `include` to filter further (e.g., keep only `reward >= 0.8` for
800
+ * rejection-sampling SFT).
801
+ *
802
+ * Realness-gated lines are dropped outright, not zeroed. SFT is imitation
803
+ * learning: unlike GRPO, where a 0 reward teaches "this trajectory was bad",
804
+ * every row here is a target to copy, so a gamed trajectory must not be in the
805
+ * file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.
806
+ *
807
+ * The exporter is fail-closed on the split, same rule as
808
+ * `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by
809
+ * default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and
810
+ * `canary` never pass the default rule. A non-training bundle that wants an
811
+ * explicit slice (e.g. a holdout-only eval bundle) names it with
812
+ * `splitFilter: ['holdout']` — explicit selection replaces the default rule.
1345
813
  */
1346
- declare function toSftRows(runs: RunRecord[], lookups: SftLookups): Promise<SftExportRow[]>;
814
+ declare function toSftRows(lines: MintedRolloutLine[], lookups: SftLookups): Promise<SftExportRow[]>;
1347
815
  declare function toSftJsonl(rows: SftExportRow[]): string;
1348
816
  interface PrmLookups {
1349
- /** Resolve the prompt text for a run. */
1350
- promptOf: (runId: string) => string | Promise<string>;
1351
- /** Resolve the trajectory step text for a (runId, spanId) pair. */
1352
- stepTextOf: (runId: string, spanId: string) => string | Promise<string>;
1353
- /** Optional: sequence of prefix span ids leading up to the divergence. */
1354
- prefixOf?: (runId: string, prefixStepIndex: number) => string[] | Promise<string[]>;
817
+ /** Resolve the prompt text for a run. */
818
+ promptOf: (runId: string) => string | Promise<string>;
819
+ /** Resolve the trajectory step text for a (runId, spanId) pair. */
820
+ stepTextOf: (runId: string, spanId: string) => string | Promise<string>;
821
+ /** Optional: sequence of prefix span ids leading up to the divergence. */
822
+ prefixOf?: (runId: string, prefixStepIndex: number) => string[] | Promise<string[]>;
1355
823
  }
1356
824
  interface PrmExportRow {
1357
- prompt: string;
1358
- /** Span ids for the steps before divergence — caller resolves text via `stepTextOf`. */
1359
- prefixSpanIds: string[];
1360
- prefixStepText: string[];
1361
- chosenStep: string;
1362
- rejectedStep: string;
1363
- chosenReward: number;
1364
- rejectedReward: number;
1365
- marginScore: number;
1366
- meta?: Record<string, unknown>;
825
+ prompt: string;
826
+ /** Span ids for the steps before divergence — caller resolves text via `stepTextOf`. */
827
+ prefixSpanIds: string[];
828
+ prefixStepText: string[];
829
+ chosenStep: string;
830
+ rejectedStep: string;
831
+ chosenReward: number;
832
+ rejectedReward: number;
833
+ marginScore: number;
834
+ meta?: Record<string, unknown>;
835
+ }
836
+ interface PrmLineContext extends RolloutLineContext {
837
+ /**
838
+ * The `maxSteps` cap the lines were minted with, if any.
839
+ *
840
+ * `mintRolloutRows` drops the MIDDLE of an over-long trajectory and leaves no
841
+ * marker on the line, so a capped trajectory is indistinguishable from a
842
+ * short one. Declaring the cap lets this exporter refuse any line sitting at
843
+ * it — a process-reward model trained on a trajectory with a hole in it
844
+ * learns credit assignment that never happened.
845
+ */
846
+ mintedWithMaxSteps?: number;
1367
847
  }
1368
848
  /**
1369
849
  * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
1370
850
  * callback resolves span text from the consumer's trace store.
851
+ *
852
+ * Every referenced run is checked against its minted line before any row is
853
+ * emitted, and the export FAILS LOUD on a trajectory that was never fully
854
+ * captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected
855
+ * side is realness-gated are dropped instead: a capture defect is the caller's
856
+ * mint configuration and must be fixed, whereas a gamed run is exactly the
857
+ * condition the gate exists to filter.
858
+ *
859
+ * `context` is REQUIRED. A two-argument call used to be accepted and produced
860
+ * rows with no gate applied at all — a `PrmTrainingTriple` carries a bare
861
+ * `chosenReward` number and nothing that says which run it came from is honest,
862
+ * so with no lines this exporter has no way to learn that its chosen step is a
863
+ * step from a run that faked its success. It now throws: fail closed, because
864
+ * the alternative is a process-reward model taught to prefer the gaming move at
865
+ * the exact step the gaming happened.
1371
866
  */
1372
- declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups): Promise<PrmExportRow[]>;
1373
- declare function toPrmJsonl(rows: PrmExportRow[]): string;
1374
- interface StepRewardJsonlRow {
1375
- runId: string;
1376
- spanId: string;
1377
- stepIndex: number;
1378
- reward: number;
1379
- determinism: 'deterministic' | 'probabilistic';
1380
- weight: number;
1381
- }
1382
- declare function stepRewardsToJsonl(stepRewards: StepReward[]): string;
1383
- declare function isTrainingRunEligible(run: RunRecord, quality: number | null | undefined, options?: TrainingRunSelectionOptions): quality is number;
1384
-
867
+ declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups, context: PrmLineContext): Promise<PrmExportRow[]>;
1385
868
  /**
1386
- * RL dataset packaging + datasheet the publishable, sellable bundle.
869
+ * Refuse to build a process-reward row from a trajectory we do not fully have.
1387
870
  *
1388
- * The format exporters (`toGrpoRows` / `toSftRows` / `toDpoRows`) already
1389
- * produce trainer-ready shapes (prime-rl GRPO, TRL DPO, conversational SFT).
1390
- * What turns that into a dataset someone can PUBLISH or BUY is the provenance
1391
- * + a datasheet: which models produced it, which prompt/agent versions, how the
1392
- * reward was derived (deterministic verifiable vs probabilistic judge — the
1393
- * credibility axis a buyer checks first), the split discipline, the reward
1394
- * distribution, the quality gates, the license, and the intended/out-of-scope
1395
- * uses. This module computes those facts from the `RunRecord[]` and renders a
1396
- * "Datasheet for Datasets" (Gebru et al. 2018) card alongside the format files.
1397
- *
1398
- * It composes the existing `rl/exporters` — it does not reimplement any trainer
1399
- * format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization
1400
- * with per-token loss masks) is a downstream Python stage that consumes the
1401
- * `messages`/`completions` this bundle emits.
871
+ * PRM training assigns credit step by step, so a missing or silently shortened
872
+ * step list is not degraded data it is data about a trajectory that never
873
+ * existed. Every condition below throws rather than filters, because each one
874
+ * means the CALLER's capture or mint configuration is wrong.
1402
875
  */
1403
-
876
+ declare function assertPrmTrainableLine(line: MintedRolloutLine, mintedWithMaxSteps?: number): void;
877
+ declare const PRM_CONTEXT_REQUIREMENT: LineContextRequirement;
878
+ declare function toPrmJsonl(rows: PrmExportRow[]): string;
879
+ interface StepRewardJsonlRow {
880
+ runId: string;
881
+ spanId: string;
882
+ stepIndex: number;
883
+ reward: number;
884
+ determinism: 'deterministic' | 'probabilistic';
885
+ weight: number;
886
+ }
887
+ declare const STEP_REWARD_CONTEXT_REQUIREMENT: LineContextRequirement;
888
+ /**
889
+ * Step-level reward rows as JSONL.
890
+ *
891
+ * `context` is REQUIRED for the same reason it is on `toDpoRows` and
892
+ * `toPrmRows`: this is a line-less input carrying a reward number. Steps
893
+ * belonging to a realness-gated run are dropped rather than zeroed — a
894
+ * per-step reward of 0 across a whole trajectory is a claim that every step was
895
+ * bad, which is a different (and false) statement from "this run's success was
896
+ * fabricated, so its step-level credit assignment is meaningless".
897
+ */
898
+ declare function stepRewardsToJsonl(stepRewards: StepReward[], context: RolloutLineContext): string;
899
+ //#endregion
900
+ //#region src/rl/dataset.d.ts
1404
901
  type RewardKind = 'deterministic' | 'probabilistic' | 'mixed';
1405
- declare const DATASET_FORMATS: readonly ["grpo", "sft", "dpo"];
902
+ declare const DATASET_FORMATS: readonly ['grpo', 'sft', 'dpo'];
1406
903
  type DatasetFormat = (typeof DATASET_FORMATS)[number];
1407
904
  declare function validateDatasetFormats(value: unknown): DatasetFormat[];
1408
905
  /** Caller-declared context — the qualitative half of the datasheet that can't
1409
906
  * be computed from records. */
1410
907
  interface RlDatasetConfig {
1411
- name: string;
1412
- version: string;
1413
- /** Product/task domain, e.g. 'legal-m&a', 'tax-1040'. */
1414
- domain: string;
1415
- /** SPDX id or a named commercial license. Required — an unlicensed dataset
1416
- * cannot be published or sold. */
1417
- license: string;
1418
- /** How the reward was produced. `kind: 'deterministic'` (a test/schema/XPath
1419
- * decided it) is the credibility signal; 'probabilistic' = LLM-judge. */
1420
- reward: {
1421
- kind: RewardKind;
1422
- source: string;
1423
- description: string;
1424
- };
1425
- intendedUse: string;
1426
- outOfScope: string;
1427
- limitations: string;
1428
- /** ISO timestamp — passed in (the substrate forbids Date.now()). */
1429
- createdAtIso: string;
1430
- /** Default: ['sft']. GRPO must be requested for multi-completion groups. */
1431
- formats?: DatasetFormat[];
1432
- /** Quality gates already run, recorded on the card for the buyer. */
1433
- qualityGates?: {
1434
- contaminationProbe?: 'passed' | 'failed' | 'not-run';
1435
- dedup?: boolean;
1436
- verifiableRewardFilter?: boolean;
1437
- };
908
+ name: string;
909
+ version: string;
910
+ /** Product/task domain, e.g. 'legal-m&a', 'tax-1040'. */
911
+ domain: string;
912
+ /** SPDX id or a named commercial license. Required — an unlicensed dataset
913
+ * cannot be published or sold. */
914
+ license: string;
915
+ /** How the reward was produced. `kind: 'deterministic'` (a test/schema/XPath
916
+ * decided it) is the credibility signal; 'probabilistic' = LLM-judge. */
917
+ reward: {
918
+ kind: RewardKind;
919
+ source: string;
920
+ description: string;
921
+ };
922
+ intendedUse: string;
923
+ outOfScope: string;
924
+ limitations: string;
925
+ /** ISO timestamp — passed in (the substrate forbids Date.now()). */
926
+ createdAtIso: string;
927
+ /** Default: ['sft']. GRPO must be requested for multi-completion groups. */
928
+ formats?: DatasetFormat[];
929
+ /** Quality gates already run, recorded on the card for the buyer. */
930
+ qualityGates?: {
931
+ contaminationProbe?: 'passed' | 'failed' | 'not-run';
932
+ dedup?: boolean;
933
+ verifiableRewardFilter?: boolean;
934
+ };
1438
935
  }
1439
936
  interface RewardStats {
1440
- n: number;
1441
- mean: number | null;
1442
- median: number | null;
1443
- min: number | null;
1444
- max: number | null;
1445
- std: number | null;
937
+ n: number;
938
+ mean: number | null;
939
+ median: number | null;
940
+ min: number | null;
941
+ max: number | null;
942
+ std: number | null;
1446
943
  }
1447
944
  interface RlDatasetStats {
1448
- records: number;
1449
- /** Records carrying an explicit task-quality score. */
1450
- scoredRecords: number;
1451
- /** Record count per split — a publishable dataset must declare its holdout. */
1452
- splits: Record<RunSplitTag, number>;
1453
- reward: RewardStats;
1454
- /** Distinct snapshot-pinned models that produced the trajectories. */
1455
- models: string[];
1456
- /** Distinct effective-prompt hashes (the agent profile/prompt versions). */
1457
- promptHashes: string[];
1458
- commitShas: string[];
1459
- totalTokens: {
1460
- input: number;
1461
- output: number;
1462
- };
1463
- totalCostUsd: number;
945
+ records: number;
946
+ /** Rollouts carrying an explicit task-quality score. */
947
+ scoredRecords: number;
948
+ /** Rollout count per split. */
949
+ splits: Record<RolloutSplit, number>;
950
+ reward: RewardStats;
951
+ /** Distinct snapshot-pinned models that produced the trajectories. */
952
+ models: string[];
953
+ /** Distinct effective-prompt hashes (the agent profile/prompt versions). */
954
+ promptHashes: string[];
955
+ commitShas: string[];
956
+ totalTokens: {
957
+ input: number;
958
+ output: number;
959
+ };
960
+ totalCostUsd: number;
961
+ /**
962
+ * Rollouts whose USD cost was never captured (`cost.usd === null`). When
963
+ * non-zero, `totalCostUsd` is a floor, not the bill — a published dataset
964
+ * must not present an unbilled run as a $0 one.
965
+ */
966
+ rolloutsWithoutCost: number;
1464
967
  }
1465
968
  interface RlDatasetManifest extends RlDatasetConfig {
1466
- formats: DatasetFormat[];
1467
- rowCounts: Partial<Record<DatasetFormat, number>>;
1468
- stats: RlDatasetStats;
969
+ formats: DatasetFormat[];
970
+ rowCounts: Partial<Record<DatasetFormat, number>>;
971
+ stats: RlDatasetStats;
1469
972
  }
1470
973
  interface RlDatasetBundle {
1471
- manifest: RlDatasetManifest;
1472
- /** Relative filename -> contents. Write these to a directory to publish. */
1473
- files: Record<string, string>;
974
+ manifest: RlDatasetManifest;
975
+ /** Relative filename -> contents. Write these to a directory to publish. */
976
+ files: Record<string, string>;
1474
977
  }
1475
978
  /**
1476
- * Package graded `RunRecord[]` into a publishable RL dataset bundle: the
979
+ * Package graded rollout lines into a publishable RL dataset bundle: the
1477
980
  * trainer-format JSONL files + a manifest + a datasheet. DPO requires
1478
981
  * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
1479
- * the records directly via the supplied lookups. Throws on an empty corpus —
982
+ * the lines directly via the supplied lookups. Throws on an empty corpus —
1480
983
  * an empty dataset must never be published.
1481
984
  */
1482
- declare function buildRlDataset(records: RunRecord[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
1483
- triples: PreferenceTriple[];
1484
- lookups: DpoLookups;
985
+ declare function buildRlDataset(lines: MintedRolloutLine[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
986
+ triples: PreferenceTriple[];
987
+ lookups: DpoLookups;
1485
988
  }): Promise<RlDatasetBundle>;
1486
989
  /** Render the "Datasheet for Datasets" card that a buyer reads. */
1487
990
  declare function datasheetToMarkdown(m: RlDatasetManifest): string;
1488
-
1489
- /**
1490
- * RL corpus — the durable, append-only accumulation of graded RunRecords that
1491
- * every eval run deposits BY DEFAULT.
1492
- *
1493
- * The dataset is the free exhaust of the normal eval process: we run evals
1494
- * constantly to get an agent production-ready, and those runs already produce
1495
- * graded trajectories. Instead of writing them to an ephemeral run dir and
1496
- * throwing them away, `appendToCorpus` accumulates them into a durable corpus;
1497
- * `buildDatasetFromCorpus` later harvests the whole corpus into a publishable
1498
- * bundle. No separate data-collection campaign — the data accrues from work we
1499
- * do anyway. This is the "best things for free by our process" layer.
1500
- *
1501
- * Trajectory text rides on the record as top-level `prompt` / `completion`
1502
- * (what the eval harnesses capture; the RunRecord validator ignores the extra
1503
- * keys). The harvest reads them directly — no trace store round-trip needed.
1504
- */
1505
-
991
+ //#endregion
992
+ //#region src/rl/corpus.d.ts
1506
993
  /** A corpus record is a RunRecord carrying the trajectory text the harness
1507
994
  * captured. `prompt`/`completion` are top-level (the validator ignores extras). */
1508
995
  type CorpusRecord = RunRecord & {
1509
- prompt?: string;
1510
- completion?: string;
996
+ prompt?: string;
997
+ completion?: string;
1511
998
  };
1512
999
  interface CorpusAppendResult {
1513
- appended: number;
1514
- /** Skipped because a record with the same runId was already in the corpus
1515
- * (idempotent appends — NOT re-run collapsing; re-runs get fresh runIds). */
1516
- skipped: number;
1517
- total: number;
1000
+ appended: number;
1001
+ /** Skipped because a record with the same runId was already in the corpus
1002
+ * (idempotent appends — NOT re-run collapsing; re-runs get fresh runIds). */
1003
+ skipped: number;
1004
+ total: number;
1518
1005
  }
1519
1006
  /**
1520
1007
  * Append graded records to the corpus (append-only JSONL). Deduplicates by
@@ -1526,12 +1013,12 @@ declare function appendToCorpus(records: CorpusRecord[], corpusPath: string): Co
1526
1013
  /** Read the full corpus. Returns [] if the corpus does not exist yet. */
1527
1014
  declare function readCorpus(corpusPath: string): CorpusRecord[];
1528
1015
  interface HarvestOptions {
1529
- /** Keep only records scoring >= this (rejection-sampling for SFT). */
1530
- minScore?: number;
1531
- /** Keep only these source splits. Held-out rows still require the explicit override below. */
1532
- splits?: RunRecord['splitTag'][];
1533
- /** Permit held-out rows in training files. Default false. */
1534
- allowHeldOutTrainingData?: boolean;
1016
+ /** Keep only records scoring >= this (rejection-sampling for SFT). */
1017
+ minScore?: number;
1018
+ /** Keep only these source splits. Held-out rows still require the explicit override below. */
1019
+ splits?: RunRecord['splitTag'][];
1020
+ /** Permit held-out rows in training files. Default false. */
1021
+ allowHeldOutTrainingData?: boolean;
1535
1022
  }
1536
1023
  /**
1537
1024
  * Harvest the accumulated corpus into a publishable RL dataset bundle. Reads
@@ -1539,2273 +1026,134 @@ interface HarvestOptions {
1539
1026
  * missing either are excluded (a graded score with no trajectory can't train).
1540
1027
  * Optionally filters by score / split. Throws (via buildRlDataset) if nothing
1541
1028
  * survives — an empty dataset must never be published.
1542
- */
1543
- declare function buildDatasetFromCorpus(corpusPath: string, config: RlDatasetConfig, opts?: HarvestOptions): Promise<RlDatasetBundle>;
1544
-
1545
- /**
1546
- * Off-policy evaluation primitives.
1547
- *
1548
- * Standard inverse-probability-weighted (IPS), self-normalized
1549
- * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
1550
- * value of a *target* policy given trajectories collected under a
1551
- * *behavior* policy. This is the canonical RL eval task: "we have last
1552
- * week's runs, we changed the policy — how would the new one do without
1553
- * re-running?"
1554
- *
1555
- * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
1556
- * & Joachims 2015 for SNIPS) but the *application* to LLM-agent
1557
- * evaluation needs care:
1558
- *
1559
- * - The "policy" is the (prompt, tool config, model snapshot) triple.
1560
- * Two policies have the same probability over an action *iff* their
1561
- * LLM call would emit the same token with the same probability —
1562
- * which is generally unknowable without the model log-probs.
1563
- * - For LLM agents, propensity scores must be supplied by the caller
1564
- * (logged in the trace, recovered from token log-probs, or estimated
1565
- * via a learned propensity model). We do NOT estimate propensity here.
1566
- * - Doubly-robust requires two outputs from a Q-function: its prediction
1567
- * for the logged action and its expectation under the target policy.
1568
- * Consumers compute these with a tabular estimate, regression fit, or
1569
- * learned reward model before constructing the trajectories.
1570
- *
1571
- * Bias / variance tradeoffs:
1572
- * - IPS: unbiased; high variance for small overlap, infinite variance
1573
- * when target has support outside behavior.
1574
- * - SNIPS: lower variance, slight bias; usually preferred in practice.
1575
- * - DR: doubly-robust — unbiased if either propensity OR Q-function is
1576
- * correct. Lowest practical variance when Q is decent. Use this.
1577
- *
1578
- * Caveat the panel will land: on the LLM-agent setting, propensity scores
1579
- * recovered from token log-probs are noisy, the action space is enormous,
1580
- * and overlap is often poor. These estimators are useful but not magic;
1581
- * complement with `replayCampaign` (exact replay where the request hashes
1582
- * match) for high-confidence answers and OPE for the gap.
1583
- */
1584
- interface OffPolicyTrajectory {
1585
- /** Stable id, for traceability through the dataset. */
1586
- runId: string;
1587
- /** Reward observed under the behavior policy (the realized outcome). */
1588
- reward: number;
1589
- /**
1590
- * Behavior-policy probability of the action that was taken. For LLM
1591
- * agents this is typically `exp(sum(token_log_probs))` over the chosen
1592
- * trajectory. Must be in (0, 1].
1593
- */
1594
- behaviorProb: number;
1595
- /**
1596
- * Target-policy probability of the same action. For replay-style
1597
- * counterfactual evaluation this is what the *new* policy would have
1598
- * assigned to the *old* trajectory. Must be in [0, 1].
1599
- */
1600
- targetProb: number;
1601
- /**
1602
- * Model-based reward prediction for the action selected by the behavior
1603
- * policy: `Q_hat(context, loggedAction)`. Supply this together with
1604
- * `vHatTarget` for contextual-bandit doubly-robust estimation.
1605
- */
1606
- qHatChosen?: number | null;
1607
- /**
1608
- * Expected model-based reward under the target policy:
1609
- * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.
1610
- * Supply this together with `qHatChosen`. For an honest evaluation, both
1611
- * values must come from a model cross-fitted or trained outside this row.
1612
- */
1613
- vHatTarget?: number | null;
1614
- /**
1615
- * @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair
1616
- * is absent, this scalar is used as both terms to preserve existing results.
1617
- * When the new pair is present, this field is ignored.
1618
- */
1619
- qHat?: number | null;
1620
- }
1621
- interface OffPolicyContributionCounts {
1622
- /** Contributions using the contextual-bandit doubly-robust formula. */
1623
- dr: number;
1624
- /** Contributions using exact IPS because no reward-model estimate was supplied. */
1625
- ipsFallback: number;
1626
- /** Contributions using the deprecated single-scalar formula. */
1627
- legacyScalar: number;
1628
- }
1629
- interface OffPolicyEstimate {
1630
- /** Estimated value of the target policy. */
1631
- value: number;
1632
- /** Standard error of the estimate. */
1633
- standardError: number;
1634
- /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
1635
- effectiveSampleSize: number;
1636
- /** Number of trajectories used. */
1637
- n: number;
1638
- /**
1639
- * Diagnostic: maximum importance weight observed. Large values (>>10x
1640
- * mean) are a red flag — variance is dominated by a few outliers.
1641
- */
1642
- maxImportanceWeight: number;
1643
- /** Populated by `doublyRobust` to expose which formula each row used. */
1644
- contributionCounts?: OffPolicyContributionCounts;
1645
- }
1646
- interface OffPolicyOptions {
1647
- /**
1648
- * Cap importance weights at this value (Ionides 2008 truncated IS) to
1649
- * trade unbiasedness for variance reduction. Default `Infinity` (no cap).
1650
- * Set e.g. `10` for stable estimates when the policies are close.
1651
- */
1652
- weightCap?: number;
1653
- /** Reward clipping range. Default `[0, 1]`. */
1654
- rewardClip?: {
1655
- low: number;
1656
- high: number;
1657
- };
1658
- }
1659
- /**
1660
- * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator
1661
- * of E[reward under target policy]. Variance scales with the spread of
1662
- * target/behavior ratios.
1663
- */
1664
- declare function inverseProbabilityWeighting(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
1665
- /**
1666
- * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at
1667
- * the cost of small bias (vanishing as N grows). The right default for
1668
- * LLM-agent evaluation where overlap is often poor.
1669
- */
1670
- declare function selfNormalizedImportanceWeighting(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
1671
- /**
1672
- * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).
1673
- *
1674
- * V_DR = (1/N) * sum_i [ v_hat_target_i
1675
- * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]
1676
- *
1677
- * Unbiased if EITHER:
1678
- * - the importance ratios are correct (IPS-style validity), OR
1679
- * - the Q-hat function is correct (model-based validity).
1680
- *
1681
- * In practice both are imperfect, but the residual bias is the *product*
1682
- * of both errors — much smaller than either alone. This is why DR is the
1683
- * default in production OPE pipelines.
1684
- *
1685
- * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither
1686
- * use the exact IPS contribution. Deprecated `qHat` rows preserve the scalar
1687
- * formula, and a complete new pair takes precedence when both forms exist.
1688
- * `contributionCounts` makes the mix explicit in the result.
1689
- * Callers must cross-fit the Q-function or train it on independent rows;
1690
- * fitting and evaluating Q on the same outcomes leaks the answer.
1691
- */
1692
- declare function doublyRobust(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
1693
- /**
1694
- * Convenience: run all three estimators and return them side-by-side.
1695
- * The recommended diagnostic — agreement across estimators is a much
1696
- * stronger signal than any single one.
1697
- */
1698
- declare function offPolicyEstimateAll(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): {
1699
- ips: OffPolicyEstimate;
1700
- snips: OffPolicyEstimate;
1701
- dr: OffPolicyEstimate;
1702
- };
1703
-
1704
- /**
1705
- * OutcomeStore — deployment outcomes attached to Run IDs.
1706
- *
1707
- * Outcomes arrive asynchronously from production telemetry after the
1708
- * eval run completed: user ratings, retention flags, conversion events,
1709
- * revenue, support-ticket rate, anything a product team can measure.
1710
- * The store is a peer to TraceStore — separate lifecycle, same runId
1711
- * foreign key.
1712
- *
1713
- * The whole point of this module is to make the meta-eval correlation
1714
- * question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.
1715
- */
1716
- interface DeploymentOutcome {
1717
- runId: string;
1718
- capturedAt: number;
1719
- /** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */
1720
- metrics: Record<string, number>;
1721
- /** Dimensions for stratified analysis — cohort, region, user_segment. */
1722
- labels?: Record<string, string>;
1723
- /** Free-form provenance (source system, pipeline version). */
1724
- source?: string;
1725
- }
1726
- interface OutcomeFilter {
1727
- runIds?: string[];
1728
- since?: number;
1729
- until?: number;
1730
- label?: {
1731
- key: string;
1732
- value: string;
1733
- };
1734
- source?: string;
1735
- }
1736
- interface OutcomeStore {
1737
- append(outcome: DeploymentOutcome): Promise<void>;
1738
- /** All outcomes attached to this run (a single run can have many — multiple
1739
- * capture windows over deployment time). */
1740
- forRun(runId: string): Promise<DeploymentOutcome[]>;
1741
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
1742
- }
1743
- declare class InMemoryOutcomeStore implements OutcomeStore {
1744
- private items;
1745
- append(outcome: DeploymentOutcome): Promise<void>;
1746
- forRun(runId: string): Promise<DeploymentOutcome[]>;
1747
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
1748
- }
1749
- interface FileSystemOutcomeStoreOptions {
1750
- dir: string;
1751
- maxBytes?: number;
1752
- }
1753
- declare class FileSystemOutcomeStore implements OutcomeStore {
1754
- private dir;
1755
- private maxBytes;
1756
- private memo?;
1757
- private loaded;
1758
- constructor(options: FileSystemOutcomeStoreOptions);
1759
- private ensureDir;
1760
- append(outcome: DeploymentOutcome): Promise<void>;
1761
- private load;
1762
- forRun(runId: string): Promise<DeploymentOutcome[]>;
1763
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
1764
- }
1765
-
1766
- /**
1767
- * Rubric predictive validity — does our eval rubric predict deployment
1768
- * outcomes?
1769
- *
1770
- * `correlationStudy` (already in this package) joins a `TraceStore` to an
1771
- * `OutcomeStore` and computes Pearson + Spearman + bootstrap CI for each
1772
- * (eval-metric, outcome-metric) pair. That answers "does X correlate with
1773
- * Y at all." `rubricPredictiveValidity` is the campaign-shaped wrapper
1774
- * around it: take a sequence of `RunRecord`s (the canonical campaign
1775
- * artifact) and a `DeploymentOutcomeStore`, join on `runId`, return a
1776
- * ranked verdict on every rubric whose dimension scores were captured in
1777
- * `outcome.raw`.
1778
- *
1779
- * The point — quoting the methodology doc — is that **without this loop
1780
- * every rubric is faith-based**. Once it's wired, you know which rubrics
1781
- * have earned their promotion power and which ones are decoration.
1782
- *
1783
- * const validity = await rubricPredictiveValidity({
1784
- * runs: lastQuarter,
1785
- * outcomes: shipFlagOutcomeStore,
1786
- * outcomeMetrics: ['revenue_lift', 'retention_30d', 'csat'],
1787
- * rubrics: ['anti_slop', 'semantic_concept', 'tool_recovery'],
1788
- * })
1789
- * for (const r of validity.ranked) {
1790
- * console.log(`${r.rubric} → ${r.bestOutcome}: ρ=${r.spearman.toFixed(2)}`)
1791
- * }
1792
- *
1793
- * The function is intentionally read-only. Use the verdict to deprecate
1794
- * decorative rubrics, re-weight composite scores, or trigger a
1795
- * recalibration sweep when predictive validity drops below a threshold.
1796
- */
1797
-
1798
- interface RubricOutcomePair {
1799
- rubric: string;
1800
- outcome: string;
1801
- n: number;
1802
- pearson: number;
1803
- spearman: number;
1804
- ci95: {
1805
- low: number;
1806
- high: number;
1807
- };
1808
- /**
1809
- * Verdict bucket. `load_bearing` ≥ 0.7, `informative` ≥ 0.4,
1810
- * `decorative` < 0.4 in absolute correlation. A negative correlation
1811
- * with a desired outcome is also `decorative` — actively misleading
1812
- * is worse than uninformative.
1813
- */
1814
- verdict: 'load_bearing' | 'informative' | 'decorative';
1815
- }
1816
- interface RubricRanking {
1817
- rubric: string;
1818
- /** Outcome metric this rubric correlated best with. */
1819
- bestOutcome: string;
1820
- spearman: number;
1821
- pearson: number;
1822
- n: number;
1823
- verdict: RubricOutcomePair['verdict'];
1824
- }
1825
- interface RubricPredictiveValidityReport {
1826
- pairs: RubricOutcomePair[];
1827
- /** Per-rubric best pair, sorted descending by |spearman|. */
1828
- ranked: RubricRanking[];
1829
- joinedSamples: number;
1830
- skippedRuns: number;
1831
- /** Rubrics that were declared but never produced a usable score. */
1832
- rubricsWithoutData: string[];
1833
- }
1834
-
1835
- /**
1836
- * HeldOutGate — first-class held-out paired-delta promotion gate.
1837
- *
1838
- * Encodes the "honesty override" pattern that lived inline in
1839
- * `~/webb/redteam/scripts/agent-eval-autoresearch.ts:138–171`.
1840
- * The optimizer's best-guess is one thing; what we should actually
1841
- * ship is another. The gate is the line between them.
1842
- *
1843
- * A candidate is promoted iff ALL three pass:
1844
- *
1845
- * 1. **Productive runs**: the candidate has at least
1846
- * `minProductiveRuns` paired observations on items where BOTH
1847
- * candidate and baseline produced a real (non-silent) score.
1848
- * 2. **Paired delta**: the lower bound of the bootstrap CI on the
1849
- * median per-item delta (candidate − baseline) on the HOLDOUT
1850
- * split is strictly greater than `pairedDeltaThreshold`.
1851
- * 3. **Overfit gap**: the candidate's gap between search-split
1852
- * score and holdout-split score is no worse (more positive)
1853
- * than the baseline's gap by more than `overfitGapThreshold`.
1854
- * "Better on search, worse on holdout" is the canonical
1855
- * overfit pattern; this catches it.
1856
- *
1857
- * The decision carries a machine-readable `rejectionCode` plus an
1858
- * `evidence` block with every number the gate looked at, so the
1859
- * downstream researcher / paper / dashboard can re-derive the
1860
- * verdict without re-running.
1861
- *
1862
- * See also:
1863
- * - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank`
1864
- * - `src/run-record.ts` for the input row schema
1865
- * - `src/reference-replay.ts` for the older, reference-replay-
1866
- * specific promotion path (still useful for replay-style evals).
1867
- */
1868
-
1869
- type HeldOutGateRejectionCode = 'few_runs' | 'missing_split_scores' | 'missing_cost' | 'negative_delta' | 'overfit_gap' | 'cost_ceiling';
1870
- interface GateEvidence {
1871
- /** Number of paired (candidate, baseline) holdout observations used. */
1872
- productiveRuns: number;
1873
- /** Candidate holdout rows with no baseline row at the same work identity. */
1874
- unpairedCandidateRuns: number;
1875
- /** Baseline holdout rows with no candidate row at the same work identity. */
1876
- unpairedBaselineRuns: number;
1877
- /** Median of paired holdout deltas, or null when there are no pairs. */
1878
- medianPairedDelta: number | null;
1879
- /** Bootstrap CI on the median paired holdout delta, if computed. */
1880
- pairedCI: {
1881
- low: number;
1882
- high: number;
1883
- } | null;
1884
- /** Wilcoxon signed-rank p-value, if computed. */
1885
- pairedPValue: number | null;
1886
- /** Mean candidate score on the search split, or null when absent. */
1887
- searchScore: number | null;
1888
- /** Mean candidate score on the holdout split, or null when absent. */
1889
- holdoutScore: number | null;
1890
- /** Candidate (search − holdout) gap, or null when either side is absent. */
1891
- overfitGap: number | null;
1892
- /** Baseline (search − holdout) gap, or null when either side is absent. */
1893
- baselineOverfitGap: number | null;
1894
- /** Median per-task USD cost across the candidate's runs. Recorded
1895
- * even when no `costPerTaskCeiling` is configured so downstream
1896
- * dashboards (intelligence.tangle.tools) can render \$/task per
1897
- * generation regardless of gating policy. */
1898
- medianCandidateCost: number | null;
1899
- /** Median per-task USD cost across the baseline runs, for
1900
- * symmetric reporting. */
1901
- medianBaselineCost: number | null;
1902
- }
1903
- interface GateDecision$1 {
1904
- /** Final promote/no-promote verdict. */
1905
- promote: boolean;
1906
- /** The candidate that was evaluated. */
1907
- candidateId: string;
1908
- /** The baseline it was compared against. */
1909
- baselineId: string;
1910
- /** Every number the gate looked at, for audit + paper export. */
1911
- evidence: GateEvidence;
1912
- /** Human-readable reason. */
1913
- reason: string;
1914
- /** Machine-readable rejection code, or null on promote. */
1915
- rejectionCode: HeldOutGateRejectionCode | null;
1916
- }
1917
-
1918
- /**
1919
- * Researcher interface — stable hook for an external autonomous-research
1920
- * agent to drive the meta-loop.
1921
- *
1922
- * Implementations live downstream (typically in a private repo that
1923
- * runs the actual LLM). This package ships only the contract + a
1924
- * `NoopResearcher` so consumers can wire the surface without being
1925
- * forced to implement every method up front.
1926
- *
1927
- * The four methods mirror the four stages of the paper "Two Loops,
1928
- * Three Roles":
1929
- *
1930
- * inspectFailures — given the observed runs, what failure modes
1931
- * are present? (data → diagnosis)
1932
- * proposeChange — given diagnosed failure modes, what
1933
- * structural changes should we try?
1934
- * (diagnosis → plan delta)
1935
- * applyChange — fold the proposed deltas into a concrete
1936
- * experiment plan against an existing baseline.
1937
- * (plan delta → executable plan)
1938
- * evaluateChange — run the plan, return runs + the gate verdict.
1939
- * (executable plan → verdict)
1940
1029
  *
1941
- * Composition is the discipline: a Researcher implementation MUST
1942
- * keep these four steps separate and inspectable. Conflating
1943
- * "diagnose + propose + run" into a single LLM call defeats the
1944
- * point of the framework you can't audit which step lied.
1945
- *
1946
- * THIS INTERFACE IS STABLE. Breaking changes require a new module
1947
- * (e.g. `Researcher2`) so existing implementations keep working.
1030
+ * `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run
1031
+ * cannot buy its way into the published bundle with its claimed score —
1032
+ * `minScore` is exactly the door a reward-hacked run would otherwise clear for
1033
+ * SFT. Unscored records are dropped before packaging: a missing label is not a
1034
+ * zero, and it is not publishable either.
1948
1035
  */
1949
-
1950
- /** A diagnosed failure mode with the run-IDs that exhibit it. */
1951
- interface FailureMode {
1952
- /** Short machine-readable code. Must be stable across runs of the
1953
- * same researcher to enable longitudinal tracking. */
1954
- code: string;
1955
- /** Human-readable description for the paper / dashboard. */
1956
- description: string;
1957
- evidence: {
1958
- /** Run IDs (from `RunRecord.runId`) where this failure mode was
1959
- * observed. */
1960
- runIds: string[];
1961
- /** Number of run samples that informed the diagnosis. */
1962
- samples: number;
1963
- };
1964
- }
1965
- /** A single steering change the researcher wants to try. */
1966
- interface SteeringChange {
1967
- kind: 'reviewer_prompt' | 'skill_add' | 'skill_remove' | 'threshold' | 'budget';
1968
- /** Implementation-specific payload. Researcher implementations
1969
- * define the schema — keep this `unknown` here to avoid coupling
1970
- * the public interface to any one researcher's internal model. */
1971
- payload: unknown;
1972
- /** Why the researcher proposed this change. Goes into the audit
1973
- * trail next to the failure-mode evidence. */
1974
- rationale: string;
1975
- /** Optional self-reported expected delta on the headline metric. */
1976
- expectedDelta?: number;
1977
- }
1978
- /** A single experiment plan, mapped onto the search/holdout splits. */
1979
- interface ExperimentPlan {
1980
- baselineCandidateId: string;
1981
- proposedCandidateId: string;
1982
- changes: SteeringChange[];
1983
- /** USD ceiling for the entire experiment. The runner must stop
1984
- * before exceeding this and report a partial result. */
1985
- evaluationBudgetUsd: number;
1986
- /** Item IDs (your dataset keys) for the search vs holdout splits. */
1987
- splits: {
1988
- search: string[];
1989
- holdout: string[];
1990
- };
1991
- }
1992
- /** Result of running a plan: every run, plus the gate verdict. */
1993
- interface ExperimentResult {
1994
- plan: ExperimentPlan;
1995
- runs: RunRecord[];
1996
- gateDecision: GateDecision$1;
1997
- }
1998
- /**
1999
- * The researcher loop. Stable, four-step, inspectable.
2000
- *
2001
- * ┌──────────┐ inspectFailures ┌──────────┐ proposeChange ┌──────────┐
2002
- * │ runs │ ─────────────────▶│ failures │ ──────────────▶│ changes │
2003
- * └──────────┘ └──────────┘ └────┬─────┘
2004
- * │
2005
- * ▼
2006
- * ┌────────────────┐ applyChange ┌────────┐
2007
- * │ ExperimentPlan │ ◀────────────│ base │
2008
- * └────────┬───────┘ └────────┘
2009
- * │
2010
- * evaluateChange ▼
2011
- * ┌────────────────┐
2012
- * │ ExperimentResult│
2013
- * └────────────────┘
2014
- */
2015
- interface Researcher {
2016
- inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
2017
- proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
2018
- applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
2019
- evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
2020
- }
2021
-
2022
- /**
2023
- * `PredictiveValidityResearcher` — concrete `Researcher` implementation
2024
- * that drives selection from outcome-anchored predictive validity.
2025
- *
2026
- * Each method:
2027
- *
2028
- * - `inspectFailures(runs)` — synthesizes failure modes from the
2029
- * bottom-quartile of `RunRecord`s on the configured proxy reward.
2030
- * - `proposeChange(failures)` — proposes steering changes that target
2031
- * the rubrics with the lowest predictive validity (decorative ones).
2032
- * Either reduce their weight in the composite, or recalibrate them.
2033
- * - `applyChange(changes, baseline)` — merges the proposed steering
2034
- * into the experiment plan.
2035
- * - `evaluateChange(plan)` — re-runs the predictive-validity check on
2036
- * the post-change runs and reports the delta.
2037
- *
2038
- * The result is a closed loop: the rubric weights drift toward the ones
2039
- * that actually predict deployment outcomes, automatically. Pair with
2040
- * `runRLCampaign` for the full auto-research story.
2041
- */
2042
-
1036
+ declare function buildDatasetFromCorpus(corpusPath: string, config: RlDatasetConfig, opts?: HarvestOptions): Promise<RlDatasetBundle>;
1037
+ //#endregion
1038
+ //#region src/rl/predictive-validity-researcher.d.ts
2043
1039
  interface PredictiveValidityResearcherOptions {
2044
- outcomes: OutcomeStore;
2045
- outcomeMetrics: string[];
2046
- /** Score threshold below which a run counts as a "failure." Default 0.5. */
2047
- failureThreshold?: number;
2048
- /** Spearman bucket below which a rubric is "decorative." Default 0.4. */
2049
- decorativeThreshold?: number;
2050
- /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
2051
- steeringNamespace?: string;
2052
- /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
2053
- rubrics?: string[];
2054
- /**
2055
- * Snapshot stash hook — called with the most recent predictive-validity
2056
- * report. Useful when a downstream system wants to log rubric drift over
2057
- * time. Default no-op.
2058
- */
2059
- onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>;
1040
+ outcomes: OutcomeStore;
1041
+ outcomeMetrics: string[];
1042
+ /** Score threshold below which a run counts as a "failure." Default 0.5. */
1043
+ failureThreshold?: number;
1044
+ /** Spearman bucket below which a rubric is "decorative." Default 0.4. */
1045
+ decorativeThreshold?: number;
1046
+ /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
1047
+ steeringNamespace?: string;
1048
+ /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
1049
+ rubrics?: string[];
1050
+ /**
1051
+ * Snapshot stash hook — called with the most recent predictive-validity
1052
+ * report. Useful when a downstream system wants to log rubric drift over
1053
+ * time. Default no-op.
1054
+ */
1055
+ onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>;
2060
1056
  }
2061
1057
  /**
2062
1058
  * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
2063
1059
  * rubrics that don't predict deployment outcomes don't earn weight.
2064
1060
  */
2065
1061
  declare class PredictiveValidityResearcher implements Researcher {
2066
- private opts;
2067
- private lastReport;
2068
- constructor(opts: PredictiveValidityResearcherOptions);
2069
- inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
2070
- proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
2071
- applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
2072
- evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
2073
- /**
2074
- * Run the predictive-validity check explicitly against a fresh RunRecord
2075
- * set. Updates the researcher's cached report so subsequent
2076
- * `proposeChange` calls have evidence to draw from.
2077
- */
2078
- runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport>;
2079
- /**
2080
- * Force-feed a predictive-validity report into the researcher state —
2081
- * useful when the consumer ran the report out-of-band and wants the
2082
- * researcher's later proposals informed by it.
2083
- */
2084
- setReport(report: RubricPredictiveValidityReport): void;
2085
- getLastReport(): RubricPredictiveValidityReport | null;
2086
- }
2087
-
2088
- /**
2089
- * Validator-output verdict — substrate primitive for "did this output pass,
2090
- * and how well?"
2091
- *
2092
- * Used by:
2093
- * - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
2094
- * - `@tangle-network/agent-runtime`Validator<Output, Verdict = DefaultVerdict>.
2095
- * Runtime keeps `Validator` because it's coupled to runtime-shaped
2096
- * `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
2097
- * itself is a substrate concept and lives here.
2098
- *
2099
- * Repo layering: agent-eval is the substrate (no upward deps). Both
2100
- * agent-runtime and agent-knowledge consume this type FROM agent-eval —
2101
- * never the other way around. See CLAUDE.md "Repo layering" for the rule.
2102
- */
2103
- /**
2104
- * Minimal verdict shape — `valid` + `score` are required; `scores` +
2105
- * `notes` are optional surface. Validators that need richer shapes
2106
- * parameterise `Validator<Output, MyVerdict>` with their own type.
2107
- *
2108
- * Need structured extras? Extend DefaultVerdict with typed fields — never
2109
- * serialize extras into `notes`.
2110
- */
2111
- interface DefaultVerdict {
2112
- /** Whether the output meets the validator's pass criteria. */
2113
- valid: boolean;
2114
- /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
2115
- score: number;
2116
- /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
2117
- scores?: Record<string, number>;
2118
- /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
2119
- notes?: string;
2120
- }
2121
-
2122
- /**
2123
- * Multi-layer verifier — ordered pipeline of verification layers.
2124
- *
2125
- * Different contract from {@link JudgeRunner} (which runs parallel
2126
- * specs against a sandbox). MultiLayerVerifier is a DAG of layers
2127
- * (install → typecheck → build → lint → serve → semantic → …) with
2128
- * dependency-based skip, per-layer findings, soft-fail semantics, and
2129
- * an aggregated `blendedScore` across all passed layers.
2130
- *
2131
- * Use when you want:
2132
- * - ordered stages where a failing upstream stage skips downstream ones
2133
- * - each stage produces rich `findings` (severity + message + evidence)
2134
- * - a single composite score across stages with per-stage weights
2135
- * - soft-fail stages whose failure doesn't abort the pipeline
2136
- *
2137
- * Use {@link JudgeRunner} when you want:
2138
- * - N independent judges running in parallel against the same artifact
2139
- * - no inter-judge dependencies
2140
- * - boolean `passed` per judge + overall
2141
- *
2142
- * Both primitives compose — JudgeRunner can be invoked as a single
2143
- * layer inside a MultiLayerVerifier if that suits the caller.
2144
- */
2145
-
2146
- type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
2147
- type Severity = 'critical' | 'major' | 'minor' | 'info';
2148
- interface Finding {
2149
- severity: Severity;
2150
- message: string;
2151
- evidence?: string;
2152
- /** Optional layer name the finding belongs to (set by the verifier if omitted). */
2153
- layer?: string;
2154
- /**
2155
- * Free-form structured payload — used by `multiToolchainLayer` to attach
2156
- * `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
2157
- * Renderers MAY interrogate; agent-eval primitives never assume shape.
2158
- */
2159
- detail?: Record<string, unknown>;
2160
- }
2161
- interface LayerResult {
2162
- layer: string;
2163
- status: LayerStatus;
2164
- /** Origin of an `error` or `timeout`. Defaults to `execution`. */
2165
- errorSource?: 'execution' | 'judge';
2166
- /** 0..1 score, optional — layers that don't produce a numeric score omit. */
2167
- score?: number;
2168
- durationMs: number;
2169
- findings: Finding[];
2170
- /** Short human-readable summary (one line). */
2171
- reason?: string;
2172
- /**
2173
- * Numeric layer-level diagnostics: error counts, warning counts,
2174
- * cyclomatic complexity, total adapter wall-time, etc. Keyed by
2175
- * diagnostic name; null = "diagnostic not applicable / not measured."
2176
- * Renderers that know the keys can display them; ones that don't,
2177
- * ignore. Free-form on purpose — consumers type the value shape in
2178
- * their own namespace.
2179
- */
2180
- diagnostics?: Record<string, number | null>;
2181
- /** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
2182
- detail?: Record<string, unknown>;
2183
- }
2184
- /** Extends the substrate verdict spine: `valid` = `allPass`; `score` is the
2185
- * complete task score or 0 when the configured scoring panel was incomplete. */
2186
- interface VerificationReport extends DefaultVerdict {
2187
- layers: LayerResult[];
2188
- passCount: number;
2189
- failCount: number;
2190
- skippedCount: number;
2191
- errorCount: number;
2192
- /** True iff the configured scoring panel completed and every layer passed. */
2193
- allPass: boolean;
2194
- /**
2195
- * Diagnostic weighted mean across contributing layers. This may represent a
2196
- * partial panel. It is 0 when no layer contributed.
2197
- */
2198
- blendedScore: number;
2199
- /**
2200
- * Complete task-quality measurement.
2201
- * Present when at least one layer produced a valid score, every other layer
2202
- * completed successfully or contributed an explicit scored failure, and no
2203
- * result is missing because of a failure, skip, error, or timeout.
2204
- * Use this field, not `blendedScore`, when creating task labels.
2205
- */
2206
- taskScore?: number;
2207
- durationMs: number;
2208
- startedAt: string;
2209
- finishedAt: string;
2210
- }
2211
-
2212
- /**
2213
- * Verifiable reward channel.
2214
- *
2215
- * For RL on coding / math / theorem-proving / structured-output tasks, the
2216
- * reward signal is *decidable* — a test passes or fails, a proof checks or
2217
- * doesn't, an output validates against a schema or doesn't. These rewards
2218
- * are dramatically more useful for RL training than LLM-judge scores
2219
- * because they don't drift, can't be Goodhart-gamed by the policy in the
2220
- * same way, and don't require a separate calibration loop.
2221
- *
2222
- * The `MultiLayerVerifier` already produces this signal — it just doesn't
2223
- * surface it in a shape that's clean enough for RL training. This module
2224
- * wraps the verifier output so consumers can:
2225
- *
2226
- * 1. Extract a clean `VerifiableReward` from a `VerificationReport`
2227
- * 2. Distinguish *deterministic* rewards (compile, test, schema) from
2228
- * *probabilistic* rewards (judge) so they can be weighted differently
2229
- * in the RL training step
2230
- * 3. Filter `RunRecord[]` to only those with a verifiable reward,
2231
- * producing the clean training set that DeepSeek-R1-style GRPO and
2232
- * AlphaProof-style search both depend on
2233
- *
2234
- * Why this matters: every credible 2025-2026 frontier RL result on coding
2235
- * agents leans on verifiable reward (DeepSeek-R1 GRPO on test pass-rate,
2236
- * o-series RL on math/code, AlphaProof on Lean kernel checking). Mixing
2237
- * judge scores into the reward signal poisons the gradient. This module
2238
- * is the seam.
2239
- */
2240
-
2241
- type VerifiableRewardSource = 'compile' | 'test' | 'schema' | 'sandbox' | 'judge' | 'composite';
2242
- interface VerifiableReward {
2243
- /** Scalar in [0, 1]. The RL training signal. */
2244
- value: number;
2245
- /** What produced the reward — different sources have different determinism. */
2246
- source: VerifiableRewardSource;
2247
- /**
2248
- * Determinism class. `'deterministic'` rewards are repeatable byte-for-byte
2249
- * given the same inputs (compile, test, schema validation, sandbox exit code).
2250
- * `'probabilistic'` rewards depend on a stochastic component (LLM judge).
2251
- * Mixing these in the same training batch without separation is a known
2252
- * footgun in production RLHF pipelines.
2253
- */
2254
- determinism: 'deterministic' | 'probabilistic';
2255
- /**
2256
- * Confidence in the reward value. For deterministic sources this is 1.0
2257
- * (the bit either flipped or didn't). For judge sources this is the
2258
- * judge-reported confidence or — when missing — a calibrated prior.
2259
- */
2260
- confidence: number;
2261
- /** The layer / judge id that produced the signal, for provenance. */
2262
- origin: string;
2263
- /**
2264
- * Per-source contribution to `value`, keyed by layer/judge id. Single-source
2265
- * rewards carry one entry (`{ [origin]: value }`); composite rewards carry
2266
- * every contributing layer's score — the anti-scalar-collapse surface RL
2267
- * consumers weight per-source instead of trusting one blended number.
2268
- */
2269
- components: Record<string, number>;
2270
- /**
2271
- * @deprecated Read `components` for per-source reward values. Kept for
2272
- * published-API compatibility: single-source rewards carry the layer's
2273
- * diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
2274
- * the same per-layer scores `components` now holds.
2275
- */
2276
- breakdown?: Record<string, number>;
2277
- }
2278
- interface VerifiableRewardExtractionOptions {
2279
- /**
2280
- * Which layers count as deterministic-reward sources. The verifier doesn't
2281
- * tag layers as "this is verifiable"; the caller declares it via this list
2282
- * (or via the layer name → source mapping). Default treats common names
2283
- * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
2284
- * `sandbox`) as deterministic.
2285
- */
2286
- deterministicLayers?: string[];
2287
- /**
2288
- * Map layer name → reward source. Defaults to a sensible string-match.
2289
- */
2290
- sourceFor?: (layerName: string) => VerifiableRewardSource;
2291
- /**
2292
- * Whether to fall back to a probabilistic (judge) reward when no
2293
- * deterministic layer produced a numeric score. Default `true`. Set to
2294
- * `false` for "deterministic-only" training pipelines that should
2295
- * discard runs without a verifiable signal.
2296
- */
2297
- fallbackToJudge?: boolean;
2298
- /**
2299
- * Default confidence for probabilistic (judge) rewards when the judge
2300
- * doesn't report one. Default `0.7`.
2301
- */
2302
- judgeConfidenceFloor?: number;
2303
- }
2304
- /**
2305
- * Extract a `VerifiableReward` from a `VerificationReport`.
2306
- *
2307
- * Strategy: prefer the deterministic layers (in order: test → compile →
2308
- * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
2309
- * true, return `null` if no signal qualifies. When multiple deterministic
2310
- * layers contribute, return a `'composite'` source with a weighted blend.
2311
- */
2312
- declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
2313
- /**
2314
- * Extract verifiable rewards from `RunRecord[]` produced via the
2315
- * `verificationReportToRunRecord` adapter (which encodes per-layer scores
2316
- * in `outcome.raw['layer.<name>']`). For records that don't carry layer
2317
- * scores, returns `null` for that record.
2318
- *
2319
- * This is the canonical bridge from "campaign-shaped artifacts" to
2320
- * "RL-training-ready reward signals": every record that has a clean
2321
- * verifiable reward becomes a training datum, every record that doesn't
2322
- * gets filtered out (or kept with `'probabilistic'` determinism for
2323
- * separate downstream handling).
2324
- */
2325
- declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
2326
- runId: string;
2327
- reward: VerifiableReward | null;
2328
- }>;
2329
- /** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
2330
- declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
2331
- run: RunRecord;
2332
- reward: VerifiableReward;
2333
- }>;
2334
-
2335
- /**
2336
- * Reward hacking / Goodhart detection.
2337
- *
2338
- * Goodhart's Law says: when a measure becomes a target, it ceases to be
2339
- * a good measure. In RLHF and agentic-RL settings this is the dominant
2340
- * failure mode — the policy learns to produce outputs that score well on
2341
- * the proxy reward (judge, rubric, test pass-rate) without producing
2342
- * the underlying capability the proxy was meant to track.
2343
- *
2344
- * Krakovna et al. (2020, "Specification Gaming Examples in AI") and the
2345
- * subsequent RLHF reward-hacking literature (Skalse et al. 2022, Kim et al.
2346
- * 2023) converge on a few diagnostic signatures:
2347
- *
2348
- * 1. **Reward divergence:** the proxy reward grows while the held-out
2349
- * ground-truth signal stagnates or drops. Predictive validity over
2350
- * time captures this.
2351
- * 2. **Distributional shift in outputs:** after RL, the policy produces
2352
- * outputs that no longer match the reference distribution — usually
2353
- * because it found a high-reward attractor that's degenerate (e.g.
2354
- * one-token responses, repetition, formatting tricks).
2355
- * 3. **Disagreement between independent rewards:** if you train on
2356
- * reward A and a held-out independent reward B drops sharply, you're
2357
- * probably hacking A.
2358
- * 4. **Calibration drift:** the verifiable / deterministic component of
2359
- * the reward is stable; the probabilistic / judge component drifts up
2360
- * while the deterministic component doesn't. The judge is being
2361
- * gamed.
2362
- *
2363
- * This module ships explicit detectors for all four signatures, plus a
2364
- * combined verdict. The output is diagnostic — actionable signals,
2365
- * not autoreject — because each signature has known false positives
2366
- * (e.g., a policy that genuinely improves can show distributional shift).
2367
- *
2368
- * Differs from `rubricPredictiveValidity` (which is a *standing* check on
2369
- * whether rubrics correlate with deployment outcomes) — this is a
2370
- * *temporal* check on whether the reward-vs-truth gap is *widening over
2371
- * time during a training run*.
2372
- */
2373
-
2374
- type RewardHackingSignal = 'reward_divergence' | 'distribution_shift' | 'reward_disagreement' | 'judge_drift';
2375
- interface RewardHackingFinding {
2376
- signal: RewardHackingSignal;
2377
- /** Severity in [0, 1]. >0.5 = strong signal. */
2378
- severity: number;
2379
- message: string;
2380
- /** Numeric evidence the consumer can render. */
2381
- detail: Record<string, number>;
2382
- }
2383
- interface RewardHackingReport {
2384
- findings: RewardHackingFinding[];
2385
- /** Signals with enough usable observations to produce a finding. */
2386
- evaluatedSignals: RewardHackingSignal[];
2387
- /**
2388
- * Composite verdict. `'insufficient_evidence'` when fewer than four scored
2389
- * runs exist; otherwise `'clean'` if every signal severity < 0.3,
2390
- * `'suspect'` if at least one ≥ 0.3 but none ≥ 0.6, and `'gaming'` if any ≥ 0.6.
2391
- */
2392
- verdict: 'insufficient_evidence' | 'clean' | 'suspect' | 'gaming';
2393
- /** Rationale for the verdict, ready to paste into an audit log. */
2394
- rationale: string[];
2395
- /** Number of runs with a usable proxy reward. */
2396
- n: number;
2397
- }
2398
- interface DetectRewardHackingInput {
2399
- /**
2400
- * Run records ordered by recency (oldest first). The detector segments
2401
- * them into prefix/suffix windows to compute "did the gap widen."
2402
- */
2403
- runs: RunRecord[];
2404
- /**
2405
- * The metric the policy was trained to optimize. Should be present on
2406
- * `outcome.raw` or `outcome.holdoutScore`. Default reads `outcome.holdoutScore`.
2407
- */
2408
- proxyOf?: (run: RunRecord) => number | null;
2409
- /**
2410
- * The held-out ground-truth metric. For RL on coding, this is typically
2411
- * test pass-rate. For RLHF, it's downstream task performance or human
2412
- * preference. For knowledge tasks, it's an independently-graded score.
2413
- */
2414
- truthOf?: (run: RunRecord) => number | null;
2415
- /**
2416
- * Independent secondary reward. Used for the `reward_disagreement`
2417
- * signal. Default uses the verifiable reward extractor (deterministic
2418
- * sources only).
2419
- */
2420
- secondaryRewardOf?: (run: RunRecord) => number | null;
2421
- /**
2422
- * Window size — how many of the most recent runs count as the "after"
2423
- * cohort. Default min(50, half the runs).
2424
- */
2425
- windowSize?: number;
2426
- /**
2427
- * Severity threshold to flag a signal. Default 0.3 (suspect) and 0.6
2428
- * (gaming).
2429
- */
2430
- thresholds?: {
2431
- suspect?: number;
2432
- gaming?: number;
2433
- };
2434
- /**
2435
- * Verifiable-reward options used for the secondary-reward fallback.
2436
- */
2437
- verifiableRewardOptions?: VerifiableRewardExtractionOptions;
2438
- }
2439
- declare function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport;
2440
-
2441
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
2442
- /** Per-million token rates for a model or endpoint not covered by package pricing. */
2443
- interface CustomTokenPricing {
2444
- /** Non-cached input tokens. */
2445
- inputUsdPerMillion: number;
2446
- /** Cache-read tokens. Falls back to the normal input rate when omitted. */
2447
- cachedInputUsdPerMillion?: number;
2448
- /** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
2449
- cacheWriteUsdPerMillion?: number;
2450
- outputUsdPerMillion: number;
2451
- }
2452
- interface ChannelRollup {
2453
- channel: CostChannel;
2454
- calls: number;
2455
- inputTokens: number;
2456
- outputTokens: number;
2457
- reasoningTokens?: number;
2458
- cachedTokens: number;
2459
- cacheWriteTokens?: number;
2460
- costUsd: number;
2461
- unpricedCalls: number;
2462
- unknownUsageCalls: number;
2463
- }
2464
- interface CostLedgerSummary {
2465
- totalCalls: number;
2466
- pendingCalls: number;
2467
- unresolvedCalls: number;
2468
- reservedCostUsd: number;
2469
- inputTokens: number;
2470
- outputTokens: number;
2471
- reasoningTokens?: number;
2472
- cachedTokens: number;
2473
- cacheWriteTokens?: number;
2474
- totalCostUsd: number;
2475
- byChannel: ChannelRollup[];
2476
- unpricedModels: string[];
2477
- fullyPriced: boolean;
2478
- usageComplete: boolean;
2479
- accountingComplete: boolean;
2480
- incompleteReasons: string[];
2481
- }
2482
-
2483
- /**
2484
- * RawProviderSink — first-class persistence for the actual HTTP-level
2485
- * request/response bodies of every LLM provider call.
2486
- *
2487
- * Why this is a separate sink from the structured `LlmSpan`:
2488
- *
2489
- * - `LlmSpan` records the *intent* — model name, messages, output text,
2490
- * usage. It's what dashboards read; it's NOT enough for forensics.
2491
- * - When a downstream consumer reports "the verifier used the wrong route"
2492
- * or "tokens look right but reasoning was missing," the only way to
2493
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
2494
- * a different `model` value than what actually answered); the raw
2495
- * response is ground truth.
2496
- *
2497
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
2498
- * matrix runner / BuilderSession sets it up automatically) and every
2499
- * request, response, and error is recorded — including retries, with the
2500
- * attempt index attached so a flaky call's full event chain is recoverable.
2501
- *
2502
- * Redaction is enforced at sink time. The default redactor strips
2503
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
2504
- * payload field whose key matches `apiKey | api_key | bearer | password |
2505
- * secret | token` (case-insensitive). Override via the sink constructor or
2506
- * the per-call `redactor`. The `redactedFields` array on the persisted
2507
- * event lets a reviewer see what was stripped without exposing the values.
2508
- */
2509
- type RawProviderDirection = 'request' | 'response' | 'error';
2510
- interface RawProviderEvent {
2511
- /** Stable id. Generated by the sink if omitted. */
2512
- eventId: string;
2513
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
2514
- runId?: string;
2515
- spanId?: string;
2516
- /**
2517
- * Logical provider name. Free-form so callers can use whatever id matches
2518
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
2519
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
2520
- */
2521
- provider: string;
2522
- model: string;
2523
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
2524
- endpoint: string;
2525
- /** Base URL used for the call (already-normalised — no trailing slash). */
2526
- baseUrl: string;
2527
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
2528
- attemptIndex: number;
2529
- direction: RawProviderDirection;
2530
- /** Unix ms. */
2531
- timestamp: number;
2532
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
2533
- durationMs?: number;
2534
- statusCode?: number;
2535
- requestHeaders?: Record<string, string>;
2536
- requestBody?: unknown;
2537
- responseHeaders?: Record<string, string>;
2538
- responseBody?: unknown;
2539
- /** Set on `direction: 'error'` events. */
2540
- errorMessage?: string;
2541
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
2542
- redactedFields: string[];
2543
- }
2544
- interface RawProviderSinkFilter {
2545
- runId?: string;
2546
- spanId?: string;
2547
- direction?: RawProviderDirection;
2548
- attemptIndex?: number;
2549
- }
2550
- interface RawProviderSink {
2551
- record(event: RawProviderEvent): Promise<void>;
2552
- /** Optional listing — implementations that durably persist (file, db) should support this. */
2553
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
2554
- /** Optional teardown for backed implementations. */
2555
- close?(): Promise<void>;
2556
- }
2557
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
2558
-
2559
- /**
2560
- * LLM client with graceful degrade.
2561
- *
2562
- * OpenAI-compatible `/v1/chat/completions` client with:
2563
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
2564
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
2565
- * - One retry at temperature 1 when a model explicitly requires it.
2566
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
2567
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
2568
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
2569
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
2570
- *
2571
- * Usage:
2572
- * const { value, result } = await callLlmJson<MyType>(
2573
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
2574
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
2575
- * )
2576
- *
2577
- * This is THE llm-calling seam for agent-eval primitives that need structured
2578
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
2579
- * that need free-form text use `callLlm` and parse output themselves.
2580
- */
2581
-
2582
- type LlmThinkingMode = 'enabled' | 'disabled';
2583
- interface LlmUsage {
2584
- promptTokens: number;
2585
- completionTokens: number;
2586
- totalTokens: number;
2587
- /** False when the provider omitted or malformed prompt/completion usage. */
2588
- captured?: boolean;
2589
- /** Reasoning-token subset of completionTokens, when reported. */
2590
- reasoningTokens?: number;
2591
- /** Proxies populate this when prompt caching is on. */
2592
- cachedPromptTokens?: number;
2593
- }
2594
- interface LlmCallResult {
2595
- /** The text content of the first choice. Empty string if none. */
2596
- content: string;
2597
- usage: LlmUsage;
2598
- /**
2599
- * Cost in USD. Uses the provider's reported cost when present, otherwise
2600
- * caller-supplied token pricing. `null` when neither is available.
2601
- */
2602
- costUsd: number | null;
2603
- /** Model name actually used (echoed from response). */
2604
- model: string;
2605
- /** Wall-clock duration of the HTTP call (last attempt, if retried). */
2606
- durationMs: number;
2607
- /**
2608
- * `finish_reason` echoed from the first choice (`stop`, `length`,
2609
- * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
2610
- * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
2611
- * (`length`) instead of treating a cut-off completion as complete. Note:
2612
- * `callLlm` does not itself reject on it — acting on this signal is the
2613
- * caller's responsibility (in-repo free-form drivers do not yet enforce it).
2614
- */
2615
- finishReason?: string | null;
2616
- /**
2617
- * True when `content.trim()` is empty. An empty completion is a silent zero
2618
- * for free-form `callLlm` callers; this flag is the signal a caller can
2619
- * inspect to fail loud rather than proceed on an empty string. `callLlm`
2620
- * surfaces it but does not throw on it.
2621
- */
2622
- contentEmpty?: boolean;
2623
- /** Raw response body. */
2624
- raw: Record<string, unknown>;
2625
- }
2626
- type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
2627
- interface LlmClientOptions {
2628
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
2629
- baseUrl?: string;
2630
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
2631
- apiKey?: string;
2632
- bearer?: string;
2633
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
2634
- authHeader?: {
2635
- name: string;
2636
- value: string;
2637
- };
2638
- /** Stable provider idempotency key, reused across retries of this logical call. */
2639
- idempotencyKey?: string;
2640
- /** Default timeout in ms. Per-call can override. */
2641
- defaultTimeoutMs?: number;
2642
- /**
2643
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
2644
- * each attempt's per-attempt timeout controller, so aborting it cancels
2645
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
2646
- * though an AbortError otherwise matches the transient patterns.
2647
- */
2648
- signal?: AbortSignal;
2649
- /**
2650
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
2651
- * Before launching each attempt the loop checks the remaining budget and
2652
- * stops retrying once it is exhausted, rather than waiting the full
2653
- * per-attempt timeout on every retry. Bounds total time independent of
2654
- * total attempts × `timeoutMs`.
2655
- */
2656
- deadlineMs?: number;
2657
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
2658
- maxRetries?: number;
2659
- /** Token rates used when the provider omits cost or package pricing does not cover the model. */
2660
- customTokenPricing?: CustomTokenPricing;
2661
- /**
2662
- * Transport for requests that declare `jsonSchema`. `native` sends
2663
- * `response_format: json_schema`; `json-object` sends the broadly supported
2664
- * JSON mode and relies on the caller to include the schema in model-visible
2665
- * instructions. Default: `native`.
2666
- */
2667
- jsonSchemaTransport?: 'native' | 'json-object';
2668
- /**
2669
- * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.
2670
- * `exact` requires the complete response content to be one JSON value.
2671
- * Default: `extract`.
2672
- */
2673
- jsonPayloadMode?: 'extract' | 'exact';
2674
- /** Default provider reasoning mode. A per-call request value takes precedence. */
2675
- thinking?: LlmThinkingMode;
2676
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
2677
- fetch?: typeof fetch;
2678
- /**
2679
- * Optional raw HTTP capture sink. When provided, every request, response,
2680
- * and error (across all retry attempts) is recorded to the sink, with auth
2681
- * headers and credential-shaped body fields redacted by default. This is
2682
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
2683
- * raw events record what actually crossed the wire.
2684
- */
2685
- rawSink?: RawProviderSink;
2686
- /**
2687
- * Logical provider id attached to raw events. When omitted, derived from
2688
- * `baseUrl` via `providerFromBaseUrl`.
2689
- */
2690
- provider?: string;
2691
- /** Trace context attached to raw events; populated by emitter-aware callers. */
2692
- traceContext?: {
2693
- runId?: string;
2694
- spanId?: string;
2695
- };
2696
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
2697
- redactor?: ProviderRedactor;
2698
- }
2699
- interface LlmRouteRequirements {
2700
- /**
2701
- * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
2702
- * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
2703
- * the public/free-tier router is a defect — the launch reviewer needs to
2704
- * know exactly which provider answered.
2705
- */
2706
- requireExplicitBaseUrl?: boolean;
2707
- /**
2708
- * Allowlist of acceptable base URLs. Strings match by prefix
2709
- * (case-insensitive); RegExps test against the full base URL.
2710
- */
2711
- allowedBaseUrls?: Array<string | RegExp>;
2712
- /** Blocklist that takes precedence over `allowedBaseUrls`. */
2713
- blockedBaseUrls?: Array<string | RegExp>;
2714
- /** Throw if no auth header / api key is configured. */
2715
- requireAuth?: boolean;
2716
- /**
2717
- * Logical provider id the configured `baseUrl` is expected to match (via
2718
- * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
2719
- */
2720
- expectedProvider?: string;
2721
- }
2722
-
2723
- /**
2724
- * FailureClusterView — groups failed runs by (failureClass, triggerTool,
2725
- * argHash-prefix) so weekly reviews can prioritize the top-N clusters.
2726
- *
2727
- * Each cluster includes: N runs, scenarios affected, representative
2728
- * error message, a proposed mitigation hint (rule → action table).
2729
- */
2730
-
2731
- interface FailureCluster {
2732
- failureClass: FailureClass;
2733
- /** Tool name when the trigger was a tool span, else undefined. */
2734
- toolName?: string;
2735
- /** First 16 chars of argHash — clusters similar args. */
2736
- argPrefix?: string;
2737
- /**
2738
- * Source dimension when the trigger was a judge span (e.g. `'format'`,
2739
- * `'safety'`, `'correctness'`). Lets cross-template aggregators
2740
- * group failures by the dimension that fired without overloading
2741
- * `argPrefix`. Optional — clusters without this field deserialize cleanly.
2742
- */
2743
- dimension?: string;
2744
- runCount: number;
2745
- scenarioIds: string[];
2746
- exampleError?: string;
2747
- exampleRunId: string;
2748
- }
2749
- interface FailureClusterReport {
2750
- clusters: FailureCluster[];
2751
- totalFailures: number;
2752
- totalRuns: number;
2753
- }
2754
-
2755
- /**
2756
- * Reporting helpers — production summaries and paper-quality figures — sit alongside `reporter.ts` rather
2757
- * than replacing it.
2758
- *
2759
- * Three artefacts:
2760
- *
2761
- * - `summaryTable` Markdown table of per-candidate means,
2762
- * 95% bootstrap CIs, BH-adjusted Wilcoxon
2763
- * p-values, and Cohen's d versus a
2764
- * comparator candidate.
2765
- * - `paretoChart` Abstract spec for a cost vs quality
2766
- * scatter, with gate decisions overlaid.
2767
- * Returns numbers + labels — caller
2768
- * chooses the plotting library.
2769
- * - `gainHistogram`
2770
- * Per-item paired holdout deltas as a
2771
- * histogram spec (bins + counts + median +
2772
- * CI). Same "data, not images" contract.
2773
- *
2774
- * The figure types are PlotSpecs — JSON-friendly, library-agnostic.
2775
- * They aren't React components and they aren't PNGs; they are
2776
- * what you'd hand to vega-lite, plotly, matplotlib, or your own
2777
- * Canvas renderer to draw the actual figure.
2778
- */
2779
-
2780
- interface SummaryTableRow {
2781
- candidateId: string;
2782
- n: number;
2783
- mean: number;
2784
- ciLow: number;
2785
- ciHigh: number;
2786
- /** BH-adjusted q-value vs comparator, or null when unavailable. */
2787
- qValue: number | null;
2788
- /** Paired Cohen's dz vs comparator, or null when the paired variance is zero. */
2789
- cohensD: number | null;
2790
- /** Matched observations used for paired comparison, or null on the comparator row. */
2791
- pairedN: number | null;
2792
- /** Candidate observations without a comparator match. */
2793
- unpairedCandidateN: number | null;
2794
- /** Comparator observations without a candidate match. */
2795
- unpairedComparatorN: number | null;
2796
- }
2797
- interface SummaryTable {
2798
- rows: SummaryTableRow[];
2799
- comparator: string | null;
2800
- split: 'search' | 'holdout';
2801
- /** Pre-rendered markdown — drop into a paper or PR. */
2802
- markdown: string;
2803
- }
2804
- interface ParetoPoint {
2805
- candidateId: string;
2806
- /** Mean USD cost per run on the chosen split. */
2807
- cost: number;
2808
- /** Mean score on the chosen split. */
2809
- quality: number;
2810
- /** Number of runs that informed this point. */
2811
- n: number;
2812
- /** Whether this candidate is on the Pareto frontier — high
2813
- * quality, low cost, no dominator. */
2814
- onFrontier: boolean;
2815
- /** Optional gate verdict for this candidate, if a `GateDecision`
2816
- * for it was passed in. */
2817
- gate?: 'promote' | 'reject';
2818
- }
2819
- interface ParetoFigureSpec {
2820
- kind: 'pareto-cost-quality';
2821
- split: 'search' | 'holdout';
2822
- points: ParetoPoint[];
2823
- axes: {
2824
- x: 'costUsd';
2825
- y: 'score';
2826
- };
2827
- }
2828
- interface GainDistributionBin {
2829
- /** Inclusive lower edge. */
2830
- lo: number;
2831
- /** Exclusive upper edge (or inclusive if it's the last bin). */
2832
- hi: number;
2833
- /** Number of pairs whose delta lands in this bin. */
2834
- count: number;
2835
- }
2836
- interface GainDistributionFigureSpec {
2837
- kind: 'gain-distribution';
2838
- candidateId: string;
2839
- comparator: string;
2840
- split: 'search' | 'holdout';
2841
- /** Number of pairs used. */
2842
- n: number;
2843
- /** Candidate rows without a comparator match. */
2844
- unpairedCandidateN: number;
2845
- /** Comparator rows without a candidate match. */
2846
- unpairedComparatorN: number;
2847
- bins: GainDistributionBin[];
2848
- median: number | null;
2849
- ci: {
2850
- low: number;
2851
- high: number;
2852
- } | null;
2853
- }
2854
- type ResearchReportDecision = 'promote' | 'hold' | 'reject' | 'equivalent' | 'needs_more_data';
2855
- interface ResearchReportOptions {
2856
- /** Human-readable report title. */
2857
- title?: string;
2858
- /** Comparator candidate id. Required for statistical decision guidance. */
2859
- comparator?: string;
2860
- /** Which split to use for the primary decision. Default 'holdout'. */
2861
- split?: 'search' | 'holdout';
2862
- /** Confidence level used by lower-level report helpers. Default 0.95. */
2863
- confidence?: number;
2864
- /** FDR threshold for q-values. Default 0.05. */
2865
- fdr?: number;
2866
- /**
2867
- * Soft floor on paired observations before issuing a directional
2868
- * promote / reject. Below this we report `needs_more_data` and surface the
2869
- * minimum detectable effect at the current N. Default 20 — chosen so the
2870
- * Wilcoxon signed-rank approximation is reasonable and so the paired
2871
- * bootstrap CI has non-degenerate coverage. Hard floor is enforced at
2872
- * `RESEARCH_REPORT_HARD_PAIR_FLOOR` (6) regardless of this value.
2873
- */
2874
- minPairs?: number;
2875
- /**
2876
- * Region of Practical Equivalence on the paired delta. When a candidate's
2877
- * paired-delta CI is fully contained in `[low, high]`, the decision is
2878
- * `equivalent` rather than `hold`. Sourced from the domain owner — there is
2879
- * no statistically-defensible default.
2880
- */
1062
+ private opts;
1063
+ private lastReport;
1064
+ constructor(opts: PredictiveValidityResearcherOptions);
1065
+ inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
1066
+ proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
1067
+ applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
1068
+ evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
1069
+ /**
1070
+ * Run the predictive-validity check explicitly against a fresh RunRecord
1071
+ * set. Updates the researcher's cached report so subsequent
1072
+ * `proposeChange` calls have evidence to draw from.
1073
+ */
1074
+ runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport>;
1075
+ /**
1076
+ * Force-feed a predictive-validity report into the researcher state —
1077
+ * useful when the consumer ran the report out-of-band and wants the
1078
+ * researcher's later proposals informed by it.
1079
+ */
1080
+ setReport(report: RubricPredictiveValidityReport): void;
1081
+ getLastReport(): RubricPredictiveValidityReport | null;
1082
+ }
1083
+ //#endregion
1084
+ //#region src/rl/rl-campaign.d.ts
1085
+ interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
1086
+ /** Preference-extraction options. Default uses paired-by-scenario-and-seed with min-margin 0.05. */
1087
+ preferences?: ExtractPreferencesOptions;
1088
+ /** Verifiable-reward extraction options. */
1089
+ verifiableReward?: VerifiableRewardExtractionOptions;
1090
+ /** Outcome store + metric names when supplied, runs `rubricPredictiveValidity` post-campaign. */
1091
+ outcomeStore?: OutcomeStore;
1092
+ outcomeMetrics?: string[];
1093
+ /** Anytime-valid sequential evaluation options. */
1094
+ sequential?: {
1095
+ alpha?: number;
1096
+ bound?: number;
2881
1097
  rope?: {
2882
- low: number;
2883
- high: number;
1098
+ low: number;
1099
+ high: number;
2884
1100
  };
2885
- /**
2886
- * Power for the minimum detectable effect (MDE) reported on each candidate.
2887
- * Default 0.8.
2888
- */
2889
- mdePower?: number;
2890
- /**
2891
- * Two-sided alpha for the MDE. Default matches `fdr` so the reported MDE
2892
- * lines up with the test the report actually runs.
2893
- */
2894
- mdeAlpha?: number;
2895
- /** Optional held-out gate decisions keyed by candidate id. */
2896
- gateDecisions?: Record<string, GateDecision$1>;
2897
- /** Optional failure clusters from failureClusterView. */
2898
- failureClusters?: FailureClusterReport;
2899
- /** Build gain histograms for these candidates. Defaults to all non-comparator candidates. */
2900
- candidateIds?: string[];
2901
- /** Deterministic bootstrap seed passed to gainHistogram and the posterior helper. */
2902
- seed?: number;
2903
- /** Report timestamp. Defaults to current time. */
2904
- generatedAt?: string;
2905
- /**
2906
- * Hash of a preregistered protocol (e.g. `signManifest({...}).contentHash`).
2907
- * Embedded verbatim in the report so the analysis can be cited as the
2908
- * preregistered one rather than a post-hoc fishing expedition.
2909
- */
2910
- preregistrationHash?: string;
2911
- }
2912
- interface ResearchReportRecommendation {
2913
- decision: ResearchReportDecision;
2914
- candidateId: string | null;
2915
- rationale: string[];
2916
- risks: string[];
2917
- nextActions: string[];
2918
- }
2919
- interface ResearchReportCandidate {
2920
- candidateId: string;
2921
- n: number;
2922
- mean: number;
2923
- ciLow: number;
2924
- ciHigh: number;
2925
- qValue: number | null;
2926
- cohensD: number | null;
2927
- meanDeltaVsComparator: number | null;
2928
- pairedN: number;
2929
- medianGain: number | null;
2930
- meanGain: number | null;
2931
- gainCi: {
2932
- low: number;
2933
- high: number;
2934
- } | null;
2935
- /**
2936
- * Bayesian-bootstrap posterior summaries on the paired mean delta.
2937
- * Dirichlet(1, ..., 1) weights represent uncertainty over the empirical
2938
- * distribution of matched deltas.
2939
- */
2940
- prGreaterThanZero: number | null;
2941
- prInRope: number | null;
2942
- /**
2943
- * Minimum detectable effect (in score units) at the candidate's paired N,
2944
- * the configured power, and the configured alpha. Standardised by the
2945
- * observed paired-delta SD and inverted via `requiredSampleSize`. Reported
2946
- * for every candidate so a `needs_more_data` verdict is actionable.
2947
- */
2948
- mde: number | null;
2949
- onParetoFrontier: boolean;
2950
- gate?: ParetoPoint['gate'];
2951
- decision: ResearchReportDecision;
2952
- decisionReason: string;
2953
- }
2954
- interface ResearchReportMethodology {
2955
- /**
2956
- * Plain-language assumptions the report depends on. Read these first when
2957
- * deciding whether the verdict is load-bearing for a launch decision.
2958
- */
2959
- assumptions: string[];
2960
- /** Tests and estimators the verdict was computed from. */
2961
- methods: string[];
2962
- /** Alternatives the author considered and why this report didn't take them. */
2963
- alternatives: string[];
2964
- /** Failure modes — when this report should NOT drive a decision. */
2965
- whenNotToApply: string[];
2966
- /** Citations for the methodological choices above. */
2967
- citations: string[];
2968
- }
2969
- interface ResearchReport {
2970
- kind: 'agent-eval-research-report';
2971
- title: string;
2972
- generatedAt: string;
2973
- split: 'search' | 'holdout';
2974
- comparator: string | null;
2975
- /**
2976
- * SHA-256 over the canonicalised set of `(runId, candidateId, split)` triples
2977
- * the report was computed from, plus the comparator and split. Stable across
2978
- * key insertion order; recomputable by the reader to verify provenance.
2979
- */
2980
- runFingerprint: string;
2981
- preregistrationHash: string | null;
2982
- rope: {
2983
- low: number;
2984
- high: number;
2985
- } | null;
2986
- executiveSummary: string[];
2987
- recommendation: ResearchReportRecommendation;
2988
- candidates: ResearchReportCandidate[];
2989
- summary: SummaryTable;
2990
- charts: {
2991
- pareto: ParetoFigureSpec;
2992
- gains: GainDistributionFigureSpec[];
2993
- };
2994
- methodology: ResearchReportMethodology;
2995
- failureClusters?: FailureClusterReport;
2996
- markdown: string;
2997
- html: string;
2998
- }
2999
-
3000
- /**
3001
- * TraceEmitter — hierarchical span builder that auto-parents using an
3002
- * internal stack. One emitter per Run; emitters do NOT share state.
3003
- *
3004
- * Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)
3005
- * return a `SpanHandle` with `.end()` / `.fail()` so callers don't
3006
- * have to thread spanIds manually. For async workflows that can't use
3007
- * the stack (e.g. fan-out parallel calls), pass `parentSpanId`
3008
- * explicitly.
3009
- */
3010
-
3011
- interface SpanHandle<S extends Span = Span> {
3012
- span: S;
3013
- end(patch?: Partial<S>): Promise<void>;
3014
- fail(error: string | Error, patch?: Partial<S>): Promise<void>;
3015
- }
3016
- interface RunCompleteHookContext {
3017
- runId: string;
3018
- emitter: TraceEmitter;
3019
- store: TraceStore;
3020
- /** Outcome the caller passed to `endRun` (undefined for `abortRun`). */
3021
- outcome?: RunOutcome$1;
3022
- /** Final run status. */
3023
- status: 'completed' | 'failed' | 'aborted';
3024
- }
3025
- type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void;
3026
- interface TraceEmitterOptions {
3027
- runId?: string;
3028
- /** Inject a clock for deterministic tests. */
3029
- now?: () => number;
3030
- /** Inject an id generator for deterministic tests. */
3031
- id?: () => string;
3032
- /**
3033
- * Hooks fired after `endRun` / `abortRun` writes the final run state.
3034
- * Designed for trace-analyst auto-execution, integrity assertions, and
3035
- * outbound notifications. Hooks run sequentially in the order supplied.
3036
- *
3037
- * By default a hook that throws is swallowed and logged as a `note` event
3038
- * on the run — auto-orchestration must not crash the underlying flow.
3039
- * Set `hookErrors: 'throw'` to propagate.
3040
- */
3041
- onRunComplete?: RunCompleteHook[];
3042
- /** `'swallow'` (default) | `'throw'`. */
3043
- hookErrors?: 'swallow' | 'throw';
3044
- }
3045
- declare class TraceEmitter {
3046
- private store;
3047
- private stack;
3048
- private _runId;
3049
- private now;
3050
- private id;
3051
- private hooks;
3052
- private hookErrors;
3053
- constructor(store: TraceStore, options?: TraceEmitterOptions);
3054
- get runId(): string;
3055
- get traceStore(): TraceStore;
3056
- /** Append a hook after construction (e.g. attach the trace analyst). */
3057
- addRunCompleteHook(hook: RunCompleteHook): void;
3058
- /**
3059
- * Begin a Run.
3060
- *
3061
- * `scenarioId` is required on the persisted Run shape — every Run downstream
3062
- * gets a non-empty scenarioId so filters and aggregations stay simple — but
3063
- * the INPUT here accepts it as optional. When omitted, startRun substitutes
3064
- * a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so
3065
- * runtime / operator / meta-eval runs that have no curated-scenario corpus
3066
- * to anchor to don't have to invent placeholder strings at the call site.
3067
- */
3068
- startRun(run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & {
3069
- scenarioId?: string;
3070
- }): Promise<Run>;
3071
- endRun(outcome?: RunOutcome$1): Promise<void>;
3072
- abortRun(reason: string): Promise<void>;
3073
- private runHooks;
3074
- span<S extends Span = Span>(init: {
3075
- kind: SpanKind;
3076
- name: string;
3077
- parentSpanId?: string;
3078
- attributes?: Record<string, unknown>;
3079
- } & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>): Promise<SpanHandle<S>>;
3080
- private handle;
3081
- private pop;
3082
- llm(init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<LlmSpan>>;
3083
- tool(init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<ToolSpan>>;
3084
- retrieval(init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<RetrievalSpan>>;
3085
- recordJudge(verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>): Promise<JudgeSpan>;
3086
- sandbox(init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<SandboxSpan>>;
3087
- emit(event: {
3088
- kind: EventKind;
3089
- spanId?: string;
3090
- payload?: Record<string, unknown>;
3091
- }): Promise<TraceEvent>;
3092
- recordBudget(entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & {
3093
- timestamp?: number;
3094
- }): Promise<BudgetLedgerEntry>;
3095
- recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact>;
3096
- /**
3097
- * Runs `fn` inside a span; auto-ends on success, auto-fails on throw.
3098
- * Returns the fn's return value. Use this for the 95% case.
3099
- */
3100
- within<T>(init: Parameters<TraceEmitter['span']>[0], fn: (handle: SpanHandle) => Promise<T>): Promise<T>;
3101
- }
3102
-
3103
- /**
3104
- * Run-completion integrity check — at end of run, verify the expected event
3105
- * types were actually captured. The point is the launch-review failure mode:
3106
- * a run *appears* successful but the raw provider events were never written,
3107
- * so a downstream reviewer can't reconstruct what happened.
3108
- *
3109
- * Pattern:
3110
- *
3111
- * const report = await assertRunCaptured(store, runId, {
3112
- * llmSpansMin: 1,
3113
- * judgeSpansMin: 1,
3114
- * rawSink: providerSink, // must have ≥ 1 event for this run
3115
- * requireRawCoverageOfLlmSpans: true, // every llm span has matching raw events
3116
- * })
3117
- * if (!report.ok) throwIfRunIncomplete(report) // or mark run failed and continue
3118
- *
3119
- * The function is read-only on the store and returns a structured report;
3120
- * the caller chooses the failure mode (throw, mark run failed, log warning).
3121
- * `throwIfRunIncomplete` is the convenient strict mode.
3122
- */
3123
-
3124
- interface RunIntegrityExpectations {
3125
- /** Minimum LLM span count. Default 0 (no requirement). */
3126
- llmSpansMin?: number;
3127
- /** Minimum judge span count. Default 0. */
3128
- judgeSpansMin?: number;
3129
- /** Minimum tool span count. Default 0. */
3130
- toolSpansMin?: number;
3131
- /**
3132
- * Raw provider sink to consult for capture verification. When present,
3133
- * the check requires at least one raw event for the run.
3134
- */
3135
- rawSink?: RawProviderSink;
3136
- /** Minimum raw provider event count. Default 0; ignored when `rawSink` absent. */
3137
- rawProviderEventsMin?: number;
3138
- /**
3139
- * Every LLM span must have at least one matching raw `request` event
3140
- * (matched by spanId). Catches the common bug where the structured span
3141
- * was emitted but the raw HTTP capture was wired to a different sink.
3142
- */
3143
- requireRawCoverageOfLlmSpans?: boolean;
3144
- /** Run outcome must be set (not null/undefined). Default false. */
3145
- requireOutcome?: boolean;
3146
- }
3147
- type RunIntegrityIssueCode = 'no_run' | 'missing_llm_spans' | 'missing_judge_spans' | 'missing_tool_spans' | 'missing_raw_events' | 'no_raw_sink' | 'orphan_llm_span' | 'missing_outcome';
3148
- interface RunIntegrityIssue {
3149
- code: RunIntegrityIssueCode;
3150
- message: string;
3151
- detail?: Record<string, unknown>;
3152
- }
3153
- interface RunIntegrityReport {
3154
- ok: boolean;
1101
+ };
1102
+ /** Trainer-format export lookups. When provided, the orchestrator builds the corresponding rows. */
1103
+ trainerExport?: {
1104
+ dpo?: DpoLookups;
1105
+ grpo?: GrpoLookups;
1106
+ sft?: SftLookups;
1107
+ };
1108
+ }
1109
+ interface RLCampaignResult {
1110
+ campaign: EvalCampaignResult;
1111
+ /** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */
1112
+ rewardSignals: Array<{
3155
1113
  runId: string;
3156
- llmSpanCount: number;
3157
- judgeSpanCount: number;
3158
- toolSpanCount: number;
3159
- rawProviderEventCount: number;
3160
- /**
3161
- * Coverage of LLM spans by raw provider events keyed on spanId.
3162
- * `total` is the number of LLM spans; `covered` is the count with at
3163
- * least one matching `request` raw event.
3164
- */
3165
- rawSpanCoverage: {
3166
- covered: number;
3167
- total: number;
3168
- };
3169
- issues: RunIntegrityIssue[];
3170
- }
3171
-
3172
- /**
3173
- * EvalCampaign opinionated matrix runner that wires the four
3174
- * capture-integrity directives by construction.
3175
- *
3176
- * The canonical benchmark shape — matrix runner → for each
3177
- * (variant, scenario, seed) start a TraceEmitter → call LLMs → end the
3178
- * run → analyze — has a bug class at the integration boundary: raw
3179
- * events not captured, route silently wrong, integrity not asserted,
3180
- * analyst never run. The directives in `SKILL.md § Capture integrity`
3181
- * are the mitigations.
3182
- *
3183
- * `EvalCampaign` is the structural fix — consumers don't wire the
3184
- * integrity surface themselves; the campaign owns it. Specifically:
3185
- *
3186
- * - calls `assertLlmRoute` once at preflight before any work runs
3187
- * - constructs a per-run `TraceStore` and `RawProviderSink` via factories
3188
- * - constructs the `TraceEmitter` with `onRunComplete: [analyst hook]`
3189
- * - hands the runner an `LlmClientOptions` pre-wired with the sink and
3190
- * trace context — the runner can't accidentally call an LLM without
3191
- * capturing the raw HTTP envelope
3192
- * - calls `assertRunCaptured` after every `endRun` and routes failures
3193
- * through a configurable policy (`throw` / `mark_failed` / `log`)
3194
- * - assembles per-run `RunRecord`s and runs `researchReport` at the end
3195
- * so the campaign artifact is launch-decision-grade by default
3196
- * - embeds the campaign fingerprint (a SHA-256 over the canonicalised
3197
- * run set) and optional `preregistrationHash` in the report
3198
- *
3199
- * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`
3200
- * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped
3201
- * lives in the campaign. This is the inversion-of-control point — consumers
3202
- * stop writing matrix runners and start writing scenario-runners.
3203
- *
3204
- * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):
3205
- *
3206
- * - Distributed/cluster execution (concurrency is local async)
3207
- * - Adaptive sampling / sequential interim looks
3208
- * - Resume from partial state across crashes
3209
- * - LLM-call retry beyond what `LlmClient` already does
3210
- */
3211
-
3212
- interface CampaignVariant<V> {
3213
- id: string;
3214
- payload: V;
3215
- }
3216
- interface CampaignScenario {
3217
- scenarioId: string;
3218
- /** Free-form metadata propagated to runs and reports. */
3219
- tags?: Record<string, string>;
3220
- }
3221
- interface CampaignRunContext<V> {
3222
- /** Stable run id. The campaign generates this; the runner does not. */
3223
- runId: string;
3224
- /** Logical experiment id (campaignId by default; overridable per-run via opts). */
3225
- experimentId: string;
3226
- variant: V;
3227
- variantId: string;
3228
- scenarioId: string;
3229
- scenarioTags: Record<string, string>;
3230
- seed: number;
3231
- splitTag: RunSplitTag;
3232
- /**
3233
- * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired
3234
- * (analyst auto-execution if configured, plus integrity check). The
3235
- * runner MUST call `emitter.startRun` before doing any work and either
3236
- * `emitter.endRun` or `emitter.abortRun` before returning.
3237
- */
3238
- emitter: TraceEmitter;
3239
- store: TraceStore;
3240
- rawSink: RawProviderSink;
3241
- /**
3242
- * Pre-wired LLM client options — `rawSink` and `traceContext` are populated
3243
- * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The
3244
- * runner can spread additional fields if needed.
3245
- */
3246
- llmOpts: LlmClientOptions;
3247
- }
3248
- interface CampaignRunOutcomeFields {
3249
- /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */
3250
- pass: boolean;
3251
- /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */
3252
- score: number;
3253
- /** Cost in USD, or null when the runner could not capture it. */
3254
- costUsd: number | null;
3255
- /** Source of the cost amount. */
3256
- costProvenance: RunCostProvenance;
3257
- tokenUsage: RunTokenUsage;
3258
- /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
3259
- model: string;
3260
- /** sha256 of the effective prompt sent to the model. */
3261
- promptHash: string;
3262
- /** sha256 of the effective config (model, temperature, tools, judges, splits). */
3263
- configHash: string;
3264
- /** Optional extra numeric metrics to land in `outcome.raw`. */
3265
- raw?: Record<string, number>;
3266
- /** Optional judge metadata when a judge was used. */
3267
- judgeMetadata?: RunJudgeMetadata;
3268
- /**
3269
- * Optional per-judge / per-dim breakdown for ensemble-judged runs.
3270
- * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.
3271
- * Single-judge or scalar-only runs leave this unset.
3272
- */
3273
- judgeScores?: JudgeScoresRecord;
3274
- /**
3275
- * Agent profile cell observed by the runner. When supplied, it overrides
3276
- * `EvalCampaignOptions.agentProfile` for this run and must match the
3277
- * outcome's `model` and `promptHash`.
3278
- */
3279
- agentProfile?: AgentProfileCell | AgentProfileCellInput;
3280
- }
3281
- /** Campaign result with the same task-failure invariant as `RunRecord`. */
3282
- type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure;
3283
- type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>;
3284
- type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log';
3285
- interface EvalCampaignOptions<V> {
3286
- /**
3287
- * Stable id for the campaign. Used as the default `experimentId` on
3288
- * every run, and folded into the campaign fingerprint.
3289
- */
3290
- campaignId: string;
3291
- variants: CampaignVariant<V>[];
3292
- scenarios: CampaignScenario[];
3293
- /** Default `[0, 1, 2]`. */
3294
- seeds?: number[];
3295
- /** Default `'holdout'` — the split that anchors a launch decision. */
3296
- splitTag?: RunSplitTag;
3297
- /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */
3298
- commitSha: string;
3299
- /**
3300
- * LLM client config. Augmented per-run with `rawSink` and `traceContext`
3301
- * before being passed to the runner. The campaign asserts this config
3302
- * matches `routeRequirements` once at preflight.
3303
- */
3304
- llmOpts: LlmClientOptions;
3305
- /**
3306
- * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail
3307
- * loud if the campaign would silently fall back to the public router or
3308
- * run unauthenticated. Override with an empty object to disable.
3309
- */
3310
- routeRequirements?: LlmRouteRequirements;
3311
- /**
3312
- * Per-run TraceStore factory. Common shape: a fresh store per run keyed
3313
- * on `runId`. Implementations that share a store across the campaign
3314
- * are valid — the campaign only writes through `emitter`.
3315
- */
3316
- storeFactory: (params: CampaignFactoryParams) => TraceStore;
3317
- /**
3318
- * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`
3319
- * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;
3320
- * otherwise required. Forensic capture is non-negotiable in a campaign
3321
- * run — pass `NoopRawProviderSink` explicitly if you want to opt out.
3322
- */
3323
- rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink;
3324
- /**
3325
- * Filesystem root for default `rawSinkFactory`. Ignored if
3326
- * `rawSinkFactory` is supplied.
3327
- */
3328
- workDir?: string;
3329
- /**
3330
- * Extra `onRunComplete` hooks the campaign appends (after its own
3331
- * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.
3332
- */
3333
- onRunComplete?: RunCompleteHook[];
3334
- /**
3335
- * Per-run integrity expectations. Defaults to:
3336
- * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.
3337
- * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.
3338
- */
3339
- integrity?: RunIntegrityExpectations;
3340
- /** Behaviour when integrity fails. Default `'mark_failed'`. */
3341
- onIntegrityFailure?: CampaignIntegrityPolicy;
3342
- /**
3343
- * Per-run runner. Receives a fully-wired context; produces an outcome
3344
- * the campaign converts into a `RunRecord`.
3345
- */
3346
- runner: CampaignRunner<V>;
3347
- /**
3348
- * If set, the campaign computes `researchReport` at the end. `comparator`
3349
- * is a `variantId`. Other fields are forwarded verbatim.
3350
- */
3351
- report?: {
3352
- comparator?: string;
3353
- } & Omit<ResearchReportOptions, 'comparator' | 'preregistrationHash' | 'generatedAt'>;
3354
- /**
3355
- * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).
3356
- * Embedded in the campaign fingerprint and the research report.
3357
- */
3358
- preregistrationHash?: string;
3359
- /** Local concurrency. Default `1` (sequential). */
3360
- concurrency?: number;
3361
- /**
3362
- * Override the time source. Tests pass a mock to make wallMs deterministic.
3363
- */
3364
- now?: () => number;
3365
- /** Override the runId generator. Tests pin this. */
3366
- runId?: (params: CampaignFactoryParams) => string;
3367
- /**
3368
- * Agent profile cell for campaign runs. Static profiles can pass an object;
3369
- * routers or variant-specific harnesses can pass a factory. The campaign
3370
- * stamps the built cell onto every `RunRecord` and rejects profile/model or
3371
- * profile/prompt contradictions.
3372
- */
3373
- agentProfile?: AgentProfileCell | AgentProfileCellInput | ((params: CampaignFactoryParams & {
3374
- variant: V;
3375
- scenarioTags: Record<string, string>;
3376
- }) => AgentProfileCell | AgentProfileCellInput | Promise<AgentProfileCell | AgentProfileCellInput>);
3377
- }
3378
- interface CampaignFactoryParams {
3379
- campaignId: string;
3380
- runId: string;
3381
- variantId: string;
3382
- scenarioId: string;
3383
- seed: number;
3384
- }
3385
- interface FailedRun {
3386
- runId: string;
3387
- variantId: string;
3388
- scenarioId: string;
3389
- seed: number;
3390
- reason: string;
3391
- error?: string;
3392
- }
3393
- interface EvalCampaignResult {
3394
- campaignId: string;
3395
- /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */
3396
- campaignFingerprint: string;
3397
- preregistrationHash: string | null;
3398
- /** Successful runs only. Failed runs land in `failedRuns`. */
3399
- runs: RunRecord[];
3400
- /** Integrity reports for every successful run. */
3401
- integrityReports: RunIntegrityReport[];
3402
- failedRuns: FailedRun[];
3403
- /** Computed when `report` is set on options. */
3404
- report?: ResearchReport;
3405
- startedAt: string;
3406
- endedAt: string;
3407
- }
3408
- declare function runEvalCampaign<V>(opts: EvalCampaignOptions<V>): Promise<EvalCampaignResult>;
3409
-
3410
- /**
3411
- * Always-valid sequential evaluation.
3412
- *
3413
- * `researchReport` assumes a single pre-specified analysis. Real
3414
- * consumers run campaigns weekly / nightly / per-PR; each new run silently
3415
- * inflates the false-discovery rate, because the BH-FDR guarantee is for
3416
- * the *first* look, not the 47th. Without time-uniform inference,
3417
- * launch-decision teams either (a) don't peek, which forfeits the cost
3418
- * advantage of stop-when-decisive, or (b) peek and pretend they didn't,
3419
- * which forfeits scientific validity.
3420
- *
3421
- * This module ships **e-value-based confidence sequences** for paired
3422
- * bounded outcomes. The methodology is the predictable plug-in betting
3423
- * martingale of Waudby-Smith & Ramdas (2024) — provably valid at *any*
3424
- * stopping time. Concretely:
3425
- *
3426
- * For paired deltas D_1, D_2, … ∈ [-c, c] with the null H_0: E[D] ≤ 0,
3427
- * a betting fraction λ_i is chosen using only D_{1..i-1} (predictable
3428
- * plug-in), and the running e-value is
3429
- *
3430
- * E_t = ∏_{i=1}^{t} (1 + λ_i · D_i)
3431
- *
3432
- * E_t is a non-negative martingale under H_0 with E[E_t] ≤ 1, so by
3433
- * Ville's inequality, P(∃ t : E_t ≥ 1/α) ≤ α — we can reject the null
3434
- * at any time without inflating the type-I error.
3435
- *
3436
- * Combined with `runEvalCampaign`, every consumer running rolling
3437
- * campaigns gains the ability to ship the moment evidence is decisive,
3438
- * stop-early on dead-on-arrival variants, and accumulate evidence across
3439
- * partial runs without spending the FDR budget. No new sweep is wasted.
3440
- *
3441
- * References:
3442
- * - Howard, S. R., Ramdas, A., McAuliffe, J., Sekhon, J. (2021).
3443
- * Time-uniform, nonparametric, nonasymptotic confidence sequences.
3444
- * Annals of Statistics, 49(2), 1055–1080.
3445
- * - Waudby-Smith, I., Ramdas, A. (2024). Estimating means of bounded
3446
- * random variables by betting. JRSS B, 86(1), 1–27.
3447
- */
3448
- type SequentialDecision = 'promote_now' | 'continue' | 'reject_now' | 'equivalent';
3449
- interface InterimReleaseConfidence {
3450
- candidates: Array<{
3451
- candidateId: string;
3452
- decision: SequentialDecision;
3453
- decisionFiredAt: number | null;
3454
- finalEvalue: number;
3455
- finalPValue: number;
3456
- pairs: number;
3457
- csLow: number;
3458
- csHigh: number;
3459
- }>;
3460
- /**
3461
- * Campaign-level recommendation: pick the strongest 'promote_now', else
3462
- * 'continue' if any candidate is still live, else 'reject_now' if every
3463
- * candidate is dead, else 'equivalent'.
3464
- */
3465
- recommendation: {
3466
- decision: SequentialDecision;
3467
- candidateId: string | null;
3468
- };
3469
- }
3470
-
3471
- /**
3472
- * `runRLCampaign` — top-level orchestrator that runs the matrix and
3473
- * produces every RL-ready artifact in one call.
3474
- *
3475
- * Wires:
3476
- * 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)
3477
- * 2. `extractVerifiableReward` over each run, separating deterministic
3478
- * from probabilistic reward sources for the trainer
3479
- * 3. `extractPreferences` to produce DPO/PPO/KTO triples
3480
- * 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)
3481
- * 5. `rubricPredictiveValidity` against an outcome store, when provided
3482
- * 6. `detectRewardHacking` as a standing hygiene check
3483
- * 7. Trainer-format export rows ready for prime-rl / TRL / verl
3484
- *
3485
- * The output `RLCampaignResult` is a single, audit-ready artifact: every
3486
- * stage's output is in there. The consumer's downstream fits in a single
3487
- * line: pass `result.preferences` to their DPO trainer, `result.grpoRows`
3488
- * to GRPO, `result.runs` plus `result.rewardSignals` to a custom RL loop.
3489
- */
3490
-
3491
- interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
3492
- /** Preference-extraction options. Default uses paired-by-scenario-and-seed with min-margin 0.05. */
3493
- preferences?: ExtractPreferencesOptions;
3494
- /** Verifiable-reward extraction options. */
3495
- verifiableReward?: VerifiableRewardExtractionOptions;
3496
- /** Outcome store + metric names — when supplied, runs `rubricPredictiveValidity` post-campaign. */
3497
- outcomeStore?: OutcomeStore;
3498
- outcomeMetrics?: string[];
3499
- /** Anytime-valid sequential evaluation options. */
3500
- sequential?: {
3501
- alpha?: number;
3502
- bound?: number;
3503
- rope?: {
3504
- low: number;
3505
- high: number;
3506
- };
3507
- };
3508
- /** Trainer-format export lookups. When provided, the orchestrator builds the corresponding rows. */
3509
- trainerExport?: {
3510
- dpo?: DpoLookups;
3511
- grpo?: GrpoLookups;
3512
- sft?: SftLookups;
3513
- };
3514
- }
3515
- interface RLCampaignResult<V> {
3516
- campaign: EvalCampaignResult;
3517
- /** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */
3518
- rewardSignals: Array<{
3519
- runId: string;
3520
- reward: VerifiableReward | null;
3521
- }>;
3522
- /** Preference extraction report. */
3523
- preferences: PreferenceExtractionReport;
3524
- /** Anytime-valid interim verdict over the paired deltas (vs comparator). */
3525
- interimConfidence: InterimReleaseConfidence | null;
3526
- /** Standing reward-hacking hygiene check. */
3527
- rewardHacking: RewardHackingReport;
3528
- /** Predictive validity, when an outcome store was supplied. */
3529
- predictiveValidity: RubricPredictiveValidityReport | null;
3530
- /** Trainer-export rows, populated only for the formats the caller requested via `trainerExport`. */
3531
- trainerRows: {
3532
- dpo?: DpoExportRow[];
3533
- grpo?: GrpoExportRow[];
3534
- sft?: SftExportRow[];
3535
- };
3536
- /**
3537
- * One-line top-level summary the consumer can log.
3538
- */
3539
- summary: string;
3540
- /**
3541
- * Convenience type-tag — consumers can branch on `result.kind`.
3542
- */
3543
- kind: 'agent-eval-rl-campaign';
3544
- unusedVariant?: V;
3545
- }
3546
- declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult<V>>;
3547
-
3548
- /**
3549
- * Pass A substrate types — `runCampaign` is the one primitive every
3550
- * eval flow composes from. Three contracts in this file:
3551
- *
3552
- * - `Scenario` input set
3553
- * - `DispatchFn` how to run one scenario → artifact
3554
- * - `CampaignResult` defined output schema (the contract downstream tools depend on)
3555
- *
3556
- * Three more lifted from earlier substrate work (re-exported):
3557
- *
3558
- * - `JudgeConfig` pluggable dimensional scorer (0.38)
3559
- * - `Mutator` optimization-loop surface mutator
3560
- * - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
3561
- *
3562
- * No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
3563
- * can build dashboards / CI gates / regression diffs against a stable schema.
3564
- */
3565
-
3566
- /** Stable identifier + kind tag for any scenario. Consumers
3567
- * extend with their per-domain payload (persona, task, requirement, ...). */
3568
- interface Scenario {
3569
- id: string;
3570
- kind: string;
3571
- tags?: string[];
3572
- /**
3573
- * Variants with the same non-empty value receive the same seed for a given
3574
- * replicate. Leave unset when every scenario should have an independent seed.
3575
- */
3576
- seedGroup?: string;
3577
- }
3578
- /** Redacted identity of a complete scenario payload retained in campaign results. */
3579
- interface CampaignScenarioIdentity extends Pick<Scenario, 'id' | 'kind'> {
3580
- scenarioDigest: `sha256:${string}`;
3581
- }
3582
- /** The canonical judge verdict shape — one declaration, shared by campaign
3583
- * judges and the multishot judge runner (which re-exports this type).
3584
- *
3585
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
3586
- * multishot runner emits 0-10. Cross-scale comparison must go through
3587
- * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
3588
- * promotion-policy) — never renormalize a producer's values in place, as
3589
- * downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
3590
- * `>= 7` gates) key on the producer's native scale. */
3591
- interface JudgeScore {
3592
- dimensions: Record<string, number>;
3593
- composite: number;
3594
- notes: string;
3595
- /** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
3596
- llmCall?: LlmCallMetadata;
3597
- /** Set when the judge itself failed (call error, unparseable output).
3598
- * `composite`/`dimensions` carry no signal — aggregators MUST exclude
3599
- * failed scores from means instead of folding them into zeros. */
3600
- failed?: true;
3601
- /** Ensemble extras (populated by `ensembleJudge`): max per-dimension
3602
- * spread across surviving judges — the inter-rater signal. */
3603
- maxDisagreement?: number;
3604
- /** Ensemble extras: judge identities whose verdict failed. */
3605
- failedJudges?: string[];
3606
- /** Ensemble extras: each surviving judge's per-dimension scores. */
3607
- perJudge?: Record<string, Record<string, number>>;
3608
- }
3609
- /** Five-valued verdict taxonomy (MOSS-paper alignment). */
3610
- type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
3611
- /** Outcome of one check that contributed to a release decision. */
3612
- type GateCheckStatus = 'pass' | 'fail' | 'not_evaluated';
3613
- interface GateContribution {
3614
- name: string;
3615
- status: GateCheckStatus;
3616
- detail: unknown;
3617
- }
3618
- interface GateResult {
3619
- decision: GateDecision;
3620
- reasons: string[];
3621
- contributingGates: GateContribution[];
3622
- delta?: number;
3623
- }
3624
- /** Token usage accumulated for a cell. Aliased to the canonical `RunTokenUsage`
3625
- * (run-record.ts, same package) so a cell maps onto a `RunRecord` for the
3626
- * backend-integrity guard with ONE source of truth — a field added to
3627
- * `RunTokenUsage` is a compile error here, not a silent drift. */
3628
- type CampaignTokenUsage = RunTokenUsage;
3629
- interface CampaignCellResult<TArtifact> {
3630
- /** Manifest that produced this cell. Resumability refuses to reuse a cell
3631
- * whose manifest differs from the current run. */
3632
- manifestHash?: string;
3633
- cellId: string;
3634
- scenarioId: string;
3635
- rep: number;
3636
- generation?: number;
3637
- artifact: TArtifact;
3638
- judgeScores: Record<string, JudgeScore>;
3639
- costUsd: number;
3640
- /** True when at least one priced receipt used the model table instead of a provider bill. */
3641
- costEstimated?: boolean;
3642
- /** Exact durable receipts required to reuse this cached result. */
3643
- costCallIds?: string[];
3644
- /** Agent-call token usage committed by `ctx.cost.runPaidCall`.
3645
- * `{ input: 0, output: 0 }` when no paid agent call was recorded. */
3646
- tokenUsage: CampaignTokenUsage;
3647
- /** Concrete model from the latest committed agent receipt. Consumed by
3648
- * `buildRunRecord` to pin the model when the declared profile uses a
3649
- * runtime-resolved sentinel. */
3650
- resolvedModel?: string;
3651
- durationMs: number;
3652
- seed: number;
3653
- cached: boolean;
3654
- /** Stage that produced `error`. Missing on successful cells. */
3655
- errorStage?: 'dispatch' | 'judge';
3656
- /** Judge that threw when `errorStage` is `judge`. */
3657
- errorJudge?: string;
3658
- error?: string;
3659
- }
3660
- interface JudgeAggregate {
3661
- mean: number;
3662
- stdev: number;
3663
- ci95: [number, number];
3664
- n: number;
3665
- }
3666
- interface ScenarioAggregate {
3667
- meanComposite: number;
3668
- ci95: [number, number];
3669
- n: number;
3670
- }
3671
- interface GenerationRecord {
3672
- generationIndex: number;
3673
- candidates: GenerationCandidate[];
3674
- promoted: string[];
3675
- }
3676
- /** One scored candidate surface in a generation. `dimensions` + `scenarios`
3677
- * let a reflective proposer ground its next proposal on WHICH
3678
- * dimensions the candidate is weakest on and WHICH scenarios it best/worst
3679
- * handled — the evidence a blind `Mutator` cannot see. */
3680
- interface GenerationCandidate {
3681
- surfaceHash: string;
3682
- /** Mean over complete task-quality scores, or null when none were produced. */
3683
- composite: number | null;
3684
- /** Descriptive interval for `composite`, or null when no score exists. */
3685
- ci95: [number, number] | null;
3686
- /** Exact surface this candidate mutated. */
3687
- parentSurfaceHash?: string;
3688
- /** Measured search-split composite of the exact parent surface. */
3689
- parentComposite?: number;
3690
- /** Candidate composite minus its parent's composite. Present only when the
3691
- * candidate completed the designed denominator. */
3692
- observedDeltaFromParent?: number;
3693
- /** Whether this candidate had a scorable result for every designed campaign
3694
- * cell and was therefore eligible for ranking, promotion, and Pareto
3695
- * selection. */
3696
- eligibleForPromotion: boolean;
3697
- /** Exact denominator receipt for selection eligibility. Scores stay
3698
- * descriptive: an incomplete candidate is retained with its observed score
3699
- * and errors instead of receiving an invented penalty. */
3700
- coverage: {
3701
- expectedCells: number;
3702
- scorableCells: number;
3703
- unscorableCells: Array<{
3704
- cellId: string;
3705
- reason: string;
3706
- }>;
3707
- };
3708
- /** Mean score per judge dimension across all cells (scenarios × reps ×
3709
- * judges that reported the dimension). */
3710
- dimensions: Record<string, number>;
3711
- /** Per-scenario composite (mean over reps + judges), plus the judge's
3712
- * free-form `notes` for that scenario — the "why it scored low" evidence a
3713
- * reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
3714
- * (which checks/lines/dimensions failed and how), NOT case-specific ground
3715
- * truth: leaking expected answers into the prompt is memorization, and the
3716
- * held-out gate would reject it anyway. `emitted` is a bounded excerpt of
3717
- * the candidate's raw output for the scenario (worst rep when reps > 1) —
3718
- * the "what it actually did" evidence; optional so trajectory capture is
3719
- * never required of a dispatch. */
3720
- scenarios: Array<{
3721
- scenarioId: string;
3722
- composite: number;
3723
- notes?: string;
3724
- emitted?: string;
3725
- }>;
3726
- /** Proposer-supplied short label for the change. Present when the proposer
3727
- * returned a `ProposedCandidate`; absent for bare-surface mutators. */
3728
- label?: string;
3729
- /** Proposer-supplied rationale — WHY this candidate was proposed. The
3730
- * "because rationale Z" the audit requires to survive to the result.
3731
- * Present when the proposer returned a `ProposedCandidate`. */
3732
- rationale?: string;
3733
- }
3734
- interface CampaignAggregates {
3735
- byJudge: Record<string, JudgeAggregate>;
3736
- byScenario: Record<string, ScenarioAggregate>;
3737
- /** Canonical campaign accounting, including worker and judge calls. */
3738
- cost: CostLedgerSummary;
3739
- /** Compatibility alias of `cost.totalCostUsd`. */
3740
- totalCostUsd: number;
3741
- /** Cells whose dispatch completed, including cells whose later judge failed. */
3742
- cellsExecuted: number;
3743
- cellsSkipped: number;
3744
- cellsCached: number;
3745
- /** All non-skipped dispatch, judge, and unclassified cell failures. */
3746
- cellsFailed: number;
3747
- /** Present on results that record failure stages. */
3748
- cellsDispatchFailed?: number;
3749
- /** Present on results that record failure stages. */
3750
- cellsJudgeFailed?: number;
3751
- /** Legacy failures whose stage was not recorded. */
3752
- cellsUnclassifiedFailed?: number;
3753
- }
3754
- interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
3755
- /** sha256(scenarios, judges, dispatch source ref, optimizer config, seed). Stable identity for reruns. */
3756
- manifestHash: string;
3757
- /** Canonical identity of the exact scenario payloads and replicate count. */
3758
- splitDigest: `sha256:${string}`;
3759
- seed: number;
3760
- /** Replicates designed for every scenario in this campaign. */
3761
- reps: number;
3762
- startedAt: string;
3763
- endedAt: string;
3764
- durationMs: number;
3765
- cells: Array<CampaignCellResult<TArtifact>>;
3766
- aggregates: CampaignAggregates;
3767
- optimization?: {
3768
- generations: GenerationRecord[];
3769
- winnerSurfaceHash?: string;
3770
- };
3771
- gate?: GateResult;
3772
- prUrl?: string;
3773
- runDir: string;
3774
- artifactsByPath: Record<string, string>;
3775
- /** Redacted identities that let consumers verify the exact scenario payloads
3776
- * without retaining customer task content in the result. */
3777
- scenarios: Array<CampaignScenarioIdentity & Pick<TScenario, 'id' | 'kind'>>;
3778
- }
3779
-
3780
- /**
3781
- * Adapters: convert measurement outputs into the canonical `RunRecord[]`
3782
- * artifact that `replayCache`, `pairedEvalueSequence`, and
3783
- * `rubricPredictiveValidity` consume. Two sources:
3784
- * - `campaignToRunRecords` — the campaign substrate's per-cell results
3785
- * (the modern path: `runCampaign` / `runImprovementLoop` → records).
3786
- * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
3787
- *
3788
- * Adapters are thin and explicit — every mandatory `RunRecord` field comes
3789
- * from a caller-supplied context (`commitSha`, `model`, `promptHash`,
3790
- * `configHash`) plus the cell's runtime data. The validator still rejects
3791
- * bare-alias model strings — the caller snapshot-pins.
3792
- */
3793
-
1114
+ reward: VerifiableReward | null;
1115
+ }>;
1116
+ /** Preference extraction report. */
1117
+ preferences: PreferenceExtractionReport;
1118
+ /** Anytime-valid interim verdict over the paired deltas (vs comparator). */
1119
+ interimConfidence: InterimReleaseConfidence | null;
1120
+ /** Standing reward-hacking hygiene check. */
1121
+ rewardHacking: RewardHackingReport;
1122
+ /** Predictive validity, when an outcome store was supplied. */
1123
+ predictiveValidity: RubricPredictiveValidityReport | null;
1124
+ /** Trainer-export rows, populated only for the formats the caller requested via `trainerExport`. */
1125
+ trainerRows: {
1126
+ dpo?: DpoExportRow[];
1127
+ grpo?: GrpoExportRow[];
1128
+ sft?: SftExportRow[];
1129
+ };
1130
+ /**
1131
+ * One-line top-level summary the consumer can log.
1132
+ */
1133
+ summary: string;
1134
+ /**
1135
+ * Convenience type-tag consumers can branch on `result.kind`.
1136
+ */
1137
+ kind: 'agent-eval-rl-campaign';
1138
+ }
1139
+ declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult>;
1140
+ //#endregion
1141
+ //#region src/rl/run-record-adapters.d.ts
3794
1142
  interface AdapterContext {
3795
- /** Logical experiment id — typically the campaign or sweep identifier. */
3796
- experimentId: string;
3797
- /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
3798
- model: string;
3799
- /** Git SHA the harness was run from. */
3800
- commitSha: string;
3801
- /** Hash of the effective prompt sent to the model. */
3802
- promptHash: string;
3803
- /** Hash of the effective config (model, temperature, tools, judges, splits). */
3804
- configHash: string;
3805
- /** Default split tag. Default `'search'`. */
3806
- splitTag?: RunSplitTag;
3807
- /** Estimated cost in USD when the source doesn't record one. */
3808
- defaultCostUsd?: number;
1143
+ /** Logical experiment id — typically the campaign or sweep identifier. */
1144
+ experimentId: string;
1145
+ /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
1146
+ model: string;
1147
+ /** Git SHA the harness was run from. */
1148
+ commitSha: string;
1149
+ /** Hash of the effective prompt sent to the model. */
1150
+ promptHash: string;
1151
+ /** Hash of the effective config (model, temperature, tools, judges, splits). */
1152
+ configHash: string;
1153
+ /** Default split tag. Default `'search'`. */
1154
+ splitTag?: RunSplitTag;
1155
+ /** Estimated cost in USD when the source doesn't record one. */
1156
+ defaultCostUsd?: number;
3809
1157
  }
3810
1158
  /**
3811
1159
  * Convert a `CampaignResult` into canonical `RunRecord[]`, one per cell.
@@ -3816,7 +1164,7 @@ interface AdapterContext {
3816
1164
  * manifest hash.
3817
1165
  */
3818
1166
  declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterContext & {
3819
- candidateId?: string;
1167
+ candidateId?: string;
3820
1168
  }): RunRecord[];
3821
1169
  /**
3822
1170
  * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
@@ -3826,43 +1174,13 @@ declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterCont
3826
1174
  * only a scored `fail` layer may produce task-failure detail.
3827
1175
  */
3828
1176
  declare function verificationReportToRunRecord(report: VerificationReport, ctx: AdapterContext & {
3829
- candidateId: string;
3830
- scenarioId: string;
1177
+ candidateId: string;
1178
+ scenarioId: string;
3831
1179
  }, opts?: {
3832
- runId?: string;
1180
+ runId?: string;
3833
1181
  }): RunRecord;
3834
-
3835
- /**
3836
- * Simulator fidelity — score a user SIMULATOR's realism against real-user
3837
- * trace distributions.
3838
- *
3839
- * Synthetic-persona evals (`PersonaConfig`-driven canonical evals, fuzz
3840
- * user-simulator objectives) stand in for real users in most of the numbers
3841
- * we publish. The standing threat is the Sim2Real gap: a simulator that is
3842
- * distributionally unlike production creates "easy mode" and silently
3843
- * inflates every score built on it. This module measures that gap from the
3844
- * SAME artifact both sides already produce — `RunRecord`s — so no new
3845
- * capture pipeline is needed:
3846
- *
3847
- * - `simFidelityReport` — per-feature Jensen-Shannon divergence between
3848
- * simulated and production record distributions, collapsed into a
3849
- * fidelity coefficient in [0,1].
3850
- * - `easyModeCheck` — the headline academic failure mode (sim inflates
3851
- * pass-rate over production) as its own named artifact.
3852
- *
3853
- * Every synthetic-persona eval result should publish its fidelity
3854
- * coefficient alongside the score — a number from an unrepresentative
3855
- * simulator is an unlabeled estimate. Wire-in points:
3856
- *
3857
- * - canonical persona evals: pass the campaign's `RunRecord`s as
3858
- * `simulated` and intake-adapter output (`contract/intake`: OTel spans,
3859
- * feedback tables, coding-agent sessions) as `production`
3860
- * - the fuzz user-sim objective: use `1 - report.fidelity` as a realism
3861
- * penalty when searching over generated personas
3862
- * - the durable corpus (`./corpus`): both sides read straight from
3863
- * `readCorpus` — tag sim vs production by `experimentId`
3864
- */
3865
-
1182
+ //#endregion
1183
+ //#region src/rl/sim-fidelity.d.ts
3866
1184
  /** Extracts a flat behavioral feature map from one record. `string` values
3867
1185
  * are categorical, `number` values are quantile-bucketed over the union of
3868
1186
  * both sides, `null` means the feature is absent on this record and is
@@ -3926,44 +1244,44 @@ declare function quantileEdges(values: number[], bucketCount?: number): number[]
3926
1244
  * `[-inf,e0)`, `[e0,e1)`, …, `[eLast,+inf)`. */
3927
1245
  declare function bucketLabel(value: number, edges: number[]): string;
3928
1246
  interface FeatureShift {
3929
- /** Category label (a string value, a numeric bucket, or `ABSENT_CATEGORY`). */
3930
- value: string;
3931
- /** Probability of this category among ALL simulated records (nulls included
3932
- * via `ABSENT_CATEGORY`, so each side's shifts sum to 1). */
3933
- pSim: number;
3934
- /** Probability among ALL production records. */
3935
- pProd: number;
1247
+ /** Category label (a string value, a numeric bucket, or `ABSENT_CATEGORY`). */
1248
+ value: string;
1249
+ /** Probability of this category among ALL simulated records (nulls included
1250
+ * via `ABSENT_CATEGORY`, so each side's shifts sum to 1). */
1251
+ pSim: number;
1252
+ /** Probability among ALL production records. */
1253
+ pProd: number;
3936
1254
  }
3937
1255
  interface FeatureDivergence {
3938
- feature: string;
3939
- /** Jensen-Shannon divergence in [0,1] for this feature. */
3940
- divergence: number;
3941
- /** Largest |pSim − pProd| categories, descending — where the sim deviates. */
3942
- topShifts: FeatureShift[];
3943
- /** Non-null observations on the simulated side. */
3944
- nSim: number;
3945
- /** Non-null observations on the production side. */
3946
- nProd: number;
1256
+ feature: string;
1257
+ /** Jensen-Shannon divergence in [0,1] for this feature. */
1258
+ divergence: number;
1259
+ /** Largest |pSim − pProd| categories, descending — where the sim deviates. */
1260
+ topShifts: FeatureShift[];
1261
+ /** Non-null observations on the simulated side. */
1262
+ nSim: number;
1263
+ /** Non-null observations on the production side. */
1264
+ nProd: number;
3947
1265
  }
3948
1266
  type FidelityVerdict = 'representative' | 'skewed' | 'insufficient-data';
3949
1267
  interface FidelityReport {
3950
- perDimension: FeatureDivergence[];
3951
- /** 1 − mean divergence over features with sufficient data. NaN when the
3952
- * verdict is 'insufficient-data' — a 0 would read as "maximally skewed"
3953
- * and silently poison downstream aggregation; check `verdict` first. */
3954
- fidelity: number;
3955
- /** Features excluded because either side had fewer than `minNPerFeature`
3956
- * non-null observations. Named, never silently dropped. */
3957
- insufficientData: string[];
3958
- /** 'representative' when fidelity >= REPRESENTATIVE_MIN_FIDELITY (0.8),
3959
- * 'skewed' below, 'insufficient-data' when no feature met minN. */
3960
- verdict: FidelityVerdict;
1268
+ perDimension: FeatureDivergence[];
1269
+ /** 1 − mean divergence over features with sufficient data. NaN when the
1270
+ * verdict is 'insufficient-data' — a 0 would read as "maximally skewed"
1271
+ * and silently poison downstream aggregation; check `verdict` first. */
1272
+ fidelity: number;
1273
+ /** Features excluded because either side had fewer than `minNPerFeature`
1274
+ * non-null observations. Named, never silently dropped. */
1275
+ insufficientData: string[];
1276
+ /** 'representative' when fidelity >= REPRESENTATIVE_MIN_FIDELITY (0.8),
1277
+ * 'skewed' below, 'insufficient-data' when no feature met minN. */
1278
+ verdict: FidelityVerdict;
3961
1279
  }
3962
1280
  interface SimFidelityOptions {
3963
- /** Feature extractor. Defaults to `defaultBehaviorFeatures`. */
3964
- features?: BehaviorFeatures;
3965
- /** Minimum non-null observations per side per feature. Default 20. */
3966
- minNPerFeature?: number;
1281
+ /** Feature extractor. Defaults to `defaultBehaviorFeatures`. */
1282
+ features?: BehaviorFeatures;
1283
+ /** Minimum non-null observations per side per feature. Default 20. */
1284
+ minNPerFeature?: number;
3967
1285
  }
3968
1286
  /**
3969
1287
  * Compare a simulator's RunRecords against production RunRecords, feature by
@@ -3973,21 +1291,21 @@ interface SimFidelityOptions {
3973
1291
  */
3974
1292
  declare function simFidelityReport(simulated: RunRecord[], production: RunRecord[], opts?: SimFidelityOptions): FidelityReport;
3975
1293
  interface EasyModeOptions {
3976
- /** A run passes when its score (holdout, else search) >= this. Default 0.5
3977
- * — matches the pass-threshold convention across the rl/ primitives. */
3978
- passThreshold?: number;
3979
- /** Pass-rate gap above which the sim is flagged inflated. Default 0.1 —
3980
- * a 10-point inflation is enough to flip most promotion gates. */
3981
- inflationTolerance?: number;
1294
+ /** A run passes when its score (holdout, else search) >= this. Default 0.5
1295
+ * — matches the pass-threshold convention across the rl/ primitives. */
1296
+ passThreshold?: number;
1297
+ /** Pass-rate gap above which the sim is flagged inflated. Default 0.1 —
1298
+ * a 10-point inflation is enough to flip most promotion gates. */
1299
+ inflationTolerance?: number;
3982
1300
  }
3983
1301
  interface EasyModeReport {
3984
- simPassRate: number;
3985
- prodPassRate: number;
3986
- /** simPassRate − prodPassRate. Positive = the simulator is easier than reality. */
3987
- gap: number;
3988
- /** True when gap > inflationTolerance: numbers measured against this
3989
- * simulator overstate production performance. */
3990
- inflated: boolean;
1302
+ simPassRate: number;
1303
+ prodPassRate: number;
1304
+ /** simPassRate − prodPassRate. Positive = the simulator is easier than reality. */
1305
+ gap: number;
1306
+ /** True when gap > inflationTolerance: numbers measured against this
1307
+ * simulator overstate production performance. */
1308
+ inflated: boolean;
3991
1309
  }
3992
1310
  /**
3993
1311
  * The headline simulator failure mode as its own named artifact: a simulator
@@ -3997,7 +1315,8 @@ interface EasyModeReport {
3997
1315
  * bias the very rate this check exists to keep honest.
3998
1316
  */
3999
1317
  declare function easyModeCheck(simulated: RunRecord[], production: RunRecord[], opts?: EasyModeOptions): EasyModeReport;
4000
-
1318
+ //#endregion
1319
+ //#region src/rl/tournament.d.ts
4001
1320
  /**
4002
1321
  * Bradley-Terry / Elo tournament evaluation.
4003
1322
  *
@@ -4025,40 +1344,40 @@ declare function easyModeCheck(simulated: RunRecord[], production: RunRecord[],
4025
1344
  * method when you have many candidates.
4026
1345
  */
4027
1346
  interface PairwiseOutcome {
4028
- /** Winner candidate id. */
4029
- winner: string;
4030
- /** Loser candidate id. */
4031
- loser: string;
4032
- /**
4033
- * Optional draw flag. When true, both candidates get half-credit
4034
- * (Bradley-Terry handles draws as half-wins for each side).
4035
- */
4036
- draw?: boolean;
4037
- /**
4038
- * Optional weight — useful if some pairwise comparisons are stronger
4039
- * signals than others (e.g. a paired test with a wider score gap is
4040
- * a more confident comparison). Default 1.
4041
- */
4042
- weight?: number;
1347
+ /** Winner candidate id. */
1348
+ winner: string;
1349
+ /** Loser candidate id. */
1350
+ loser: string;
1351
+ /**
1352
+ * Optional draw flag. When true, both candidates get half-credit
1353
+ * (Bradley-Terry handles draws as half-wins for each side).
1354
+ */
1355
+ draw?: boolean;
1356
+ /**
1357
+ * Optional weight — useful if some pairwise comparisons are stronger
1358
+ * signals than others (e.g. a paired test with a wider score gap is
1359
+ * a more confident comparison). Default 1.
1360
+ */
1361
+ weight?: number;
4043
1362
  }
4044
1363
  interface BradleyTerryRating {
4045
- candidateId: string;
4046
- /** Latent strength θ ≥ 0 from the BT MLE. */
4047
- strength: number;
4048
- /** Log-strength = log(θ) — interpretable on a linear scale. */
4049
- logStrength: number;
4050
- /** Number of pairwise comparisons this candidate appears in. */
4051
- n: number;
4052
- /** Win count (+ 0.5 per draw). */
4053
- wins: number;
1364
+ candidateId: string;
1365
+ /** Latent strength θ ≥ 0 from the BT MLE. */
1366
+ strength: number;
1367
+ /** Log-strength = log(θ) — interpretable on a linear scale. */
1368
+ logStrength: number;
1369
+ /** Number of pairwise comparisons this candidate appears in. */
1370
+ n: number;
1371
+ /** Win count (+ 0.5 per draw). */
1372
+ wins: number;
4054
1373
  }
4055
1374
  interface BradleyTerryFit {
4056
- ratings: BradleyTerryRating[];
4057
- /** Iterations of the MM algorithm before convergence. */
4058
- iterations: number;
4059
- /** Final maximum |θ_new - θ_old| / θ_old. */
4060
- finalDelta: number;
4061
- converged: boolean;
1375
+ ratings: BradleyTerryRating[];
1376
+ /** Iterations of the MM algorithm before convergence. */
1377
+ iterations: number;
1378
+ /** Final maximum |θ_new - θ_old| / θ_old. */
1379
+ finalDelta: number;
1380
+ converged: boolean;
4062
1381
  }
4063
1382
  /**
4064
1383
  * Bradley-Terry MLE via Hunter's MM algorithm.
@@ -4070,9 +1389,9 @@ interface BradleyTerryFit {
4070
1389
  * offset is unobservable in BT — only differences are identified).
4071
1390
  */
4072
1391
  declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
4073
- tolerance?: number;
4074
- maxIterations?: number;
4075
- smoothing?: number;
1392
+ tolerance?: number;
1393
+ maxIterations?: number;
1394
+ smoothing?: number;
4076
1395
  }): BradleyTerryFit;
4077
1396
  /**
4078
1397
  * Online Elo updates. Use when comparisons arrive over time and you want
@@ -4083,14 +1402,14 @@ declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
4083
1402
  * the caller can log per-comparison rating changes.
4084
1403
  */
4085
1404
  interface EloOptions {
4086
- /** Default rating for unseen candidates. Default 1500. */
4087
- defaultRating?: number;
4088
- /** K-factor controls the step size. Default 32 (FIDE-ish). */
4089
- kFactor?: number;
1405
+ /** Default rating for unseen candidates. Default 1500. */
1406
+ defaultRating?: number;
1407
+ /** K-factor controls the step size. Default 32 (FIDE-ish). */
1408
+ kFactor?: number;
4090
1409
  }
4091
1410
  declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseOutcome, opts?: EloOptions): {
4092
- winnerDelta: number;
4093
- loserDelta: number;
1411
+ winnerDelta: number;
1412
+ loserDelta: number;
4094
1413
  };
4095
1414
  /**
4096
1415
  * Build pairwise outcomes from the campaign artifact: for every scenario
@@ -4099,18 +1418,19 @@ declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseO
4099
1418
  * pairwise judge call.
4100
1419
  */
4101
1420
  interface BuildPairwiseFromCampaignInput {
4102
- runs: Array<{
4103
- candidateId: string;
4104
- /** Stable identifier for the matching unit (typically scenarioId). */
4105
- matchKey: string;
4106
- score: number;
4107
- }>;
4108
- /**
4109
- * Tied-score margin. Below this, the comparison is a draw. Default 0
4110
- * (no ties).
4111
- */
4112
- drawMargin?: number;
1421
+ runs: Array<{
1422
+ candidateId: string;
1423
+ /** Stable identifier for the matching unit (typically scenarioId). */
1424
+ matchKey: string;
1425
+ score: number;
1426
+ }>;
1427
+ /**
1428
+ * Tied-score margin. Below this, the comparison is a draw. Default 0
1429
+ * (no ties).
1430
+ */
1431
+ drawMargin?: number;
4113
1432
  }
4114
1433
  declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
4115
-
4116
- export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type AdversarialMutation, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DatasetFormat, type DeploymentOutcome, type DetectRewardHackingInput, type DpoExportRow, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type GrpoExportRow, type GrpoLookups, type HarvestOptions, InMemoryOutcomeStore, type OffPolicyContributionCounts, type OffPolicyEstimate, type OffPolicyOptions, type OffPolicyTrajectory, type OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type TrainingRunSelectionOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, appendToCorpus, applyEloUpdate, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, isTrainingRunEligible, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };
1434
+ //#endregion
1435
+ export { ABSENT_CATEGORY, AdaptationCurve, AdaptationPoint, AdaptationRunner, AdapterContext, AdversarialMutation, BehaviorFeatures, BradleyTerryFit, BradleyTerryRating, BuildPairwiseFromCampaignInput, CellObservation, CompareCurvesResult, ComputeBestOfNOptions, ComputeBestOfNResult, ComputeCurve, ComputeCurveBudget, ComputeCurvePoint, ContaminationProbeInput, ContaminationProbeOptions, ContaminationProbeReport, CorpusAppendResult, CorpusRecord, CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, DPO_CONTEXT_REQUIREMENT, DatasetFormat, type DeploymentOutcome, DetectRewardHackingInput, DpoExportRow, DpoLineContext, DpoLookups, EasyModeOptions, EasyModeReport, EloOptions, ExtractPreferencesOptions, ExtractStepRewardsOptions, FeatureDivergence, FeatureShift, FidelityReport, FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, GrpoExportRow, GrpoLookups, HarvestOptions, InMemoryOutcomeStore, OffPolicyContributionCounts, OffPolicyEstimate, OffPolicyOptions, OffPolicyTrajectory, type OutcomeStore, PRM_CONTEXT_REQUIREMENT, PairwiseOutcome, ParetoPointInput, PredictiveValidityResearcher, PredictiveValidityResearcherOptions, PreferenceExtractionReport, PreferenceStrategy, PreferenceTriple, PrmExportRow, PrmLineContext, PrmLookups, PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, RLCampaignResult, RewardHackingFinding, RewardHackingReport, RewardHackingSignal, RewardKind, RewardStats, RlDatasetBundle, RlDatasetConfig, RlDatasetManifest, RlDatasetStats, type RolloutLineContext, RunAdaptationCurveOptions, RunComputeCurveOptions, RunRLCampaignOptions, RunwiseStepSummary, STEP_REWARD_CONTEXT_REQUIREMENT, ScenarioPerturbation, ScenarioPerturbationKind, SelfConsistencyOptions, SelfConsistencyResult, SftExportRow, SftLookups, SimFidelityOptions, StepReward, StepRewardJsonlRow, StepScorer, ThompsonCurriculumOptions, TrainingLineSelectionOptions, VarianceCurriculumOptions, VerifiableReward, VerifiableRewardExtractionOptions, VerifiableRewardSource, appendToCorpus, applyEloUpdate, assertPrmTrainableLine, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };
1436
+ //# sourceMappingURL=rl.d.ts.map