@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
package/dist/reporting.js CHANGED
@@ -1,51 +1,6 @@
1
- import {
2
- assertReleaseConfidence,
3
- bootstrapCi,
4
- evaluateReleaseConfidence,
5
- judgeReplayGate,
6
- renderReleaseReport
7
- } from "./chunk-JQSF5DQT.js";
8
- import {
9
- rubricPredictiveValidity
10
- } from "./chunk-TQ7LNKZ3.js";
11
- import {
12
- evaluateInterimReleaseConfidence,
13
- pairedEvalueSequence
14
- } from "./chunk-MAZ26DC7.js";
15
- import {
16
- RESEARCH_REPORT_HARD_PAIR_FLOOR,
17
- gainHistogram,
18
- paretoChart,
19
- researchReport,
20
- summaryTable
21
- } from "./chunk-TJVT4QFF.js";
22
- import "./chunk-7FO3TNPI.js";
23
- import {
24
- benjaminiHochberg,
25
- pairedBootstrap,
26
- wilcoxonSignedRank
27
- } from "./chunk-ZHTZ4EYI.js";
28
- import "./chunk-56TAVBOK.js";
29
- import "./chunk-MA6HLL3S.js";
30
- import "./chunk-OIUOT4QD.js";
31
- import "./chunk-ONWEPEDO.js";
32
- import "./chunk-PZ5AY32C.js";
33
- export {
34
- RESEARCH_REPORT_HARD_PAIR_FLOOR,
35
- assertReleaseConfidence,
36
- benjaminiHochberg,
37
- bootstrapCi,
38
- evaluateInterimReleaseConfidence,
39
- evaluateReleaseConfidence,
40
- gainHistogram,
41
- judgeReplayGate,
42
- pairedBootstrap,
43
- pairedEvalueSequence,
44
- paretoChart,
45
- renderReleaseReport,
46
- researchReport,
47
- rubricPredictiveValidity,
48
- summaryTable,
49
- wilcoxonSignedRank
50
- };
51
- //# sourceMappingURL=reporting.js.map
1
+ import { N as wilcoxonSignedRank, t as benjaminiHochberg, v as pairedBootstrap } from "./statistics-CnnxdpOg.js";
2
+ import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Cz9NKH39.js";
3
+ import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BNs5nmXI.js";
4
+ import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
5
+ import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-B3xmbmS1.js";
6
+ export { RESEARCH_REPORT_HARD_PAIR_FLOOR, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, renderReleaseReport, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
@@ -0,0 +1,314 @@
1
+ import { a as RunRecord, b as AgentProfileCellInput, c as RunTaskFailure, n as RunCostProvenance, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage, y as AgentProfileCell } from "./run-record-CnZu_gjl.js";
2
+ import { l as RawProviderSink } from "./raw-provider-sink-BU29Sh8h.js";
3
+ import { s as TraceStore } from "./store-CT9YIIve.js";
4
+ import { i as TraceEmitter, t as RunCompleteHook } from "./emitter-DGQGoLyj.js";
5
+ import { a as RunIntegrityReport, n as RunIntegrityExpectations } from "./integrity-rmVhXWA7.js";
6
+ import { o as LlmClientOptions, u as LlmRouteRequirements } from "./llm-client-B_nIBlYo.js";
7
+ import { b as GateDecision, d as ResearchReportOptions, s as ResearchReport } from "./summary-report-Cj9gdw4i.js";
8
+ //#region src/eval-campaign.d.ts
9
+ interface CampaignVariant<V> {
10
+ id: string;
11
+ payload: V;
12
+ }
13
+ interface CampaignScenario {
14
+ scenarioId: string;
15
+ /** Free-form metadata propagated to runs and reports. */
16
+ tags?: Record<string, string>;
17
+ }
18
+ interface CampaignRunContext<V> {
19
+ /** Stable run id. The campaign generates this; the runner does not. */
20
+ runId: string;
21
+ /** Logical experiment id (campaignId by default; overridable per-run via opts). */
22
+ experimentId: string;
23
+ variant: V;
24
+ variantId: string;
25
+ scenarioId: string;
26
+ scenarioTags: Record<string, string>;
27
+ seed: number;
28
+ splitTag: RunSplitTag;
29
+ /**
30
+ * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired
31
+ * (analyst auto-execution if configured, plus integrity check). The
32
+ * runner MUST call `emitter.startRun` before doing any work and either
33
+ * `emitter.endRun` or `emitter.abortRun` before returning.
34
+ */
35
+ emitter: TraceEmitter;
36
+ store: TraceStore;
37
+ rawSink: RawProviderSink;
38
+ /**
39
+ * Pre-wired LLM client options — `rawSink` and `traceContext` are populated
40
+ * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The
41
+ * runner can spread additional fields if needed.
42
+ */
43
+ llmOpts: LlmClientOptions;
44
+ }
45
+ interface CampaignRunOutcomeFields {
46
+ /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */
47
+ pass: boolean;
48
+ /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */
49
+ score: number;
50
+ /** Cost in USD, or null when the runner could not capture it. */
51
+ costUsd: number | null;
52
+ /** Source of the cost amount. */
53
+ costProvenance: RunCostProvenance;
54
+ tokenUsage: RunTokenUsage;
55
+ /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
56
+ model: string;
57
+ /** sha256 of the effective prompt sent to the model. */
58
+ promptHash: string;
59
+ /** sha256 of the effective config (model, temperature, tools, judges, splits). */
60
+ configHash: string;
61
+ /** Optional extra numeric metrics to land in `outcome.raw`. */
62
+ raw?: Record<string, number>;
63
+ /** Optional judge metadata when a judge was used. */
64
+ judgeMetadata?: RunJudgeMetadata;
65
+ /**
66
+ * Optional per-judge / per-dim breakdown for ensemble-judged runs.
67
+ * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.
68
+ * Single-judge or scalar-only runs leave this unset.
69
+ */
70
+ judgeScores?: JudgeScoresRecord;
71
+ /**
72
+ * Agent profile cell observed by the runner. When supplied, it overrides
73
+ * `EvalCampaignOptions.agentProfile` for this run and must match the
74
+ * outcome's `model` and `promptHash`.
75
+ */
76
+ agentProfile?: AgentProfileCell | AgentProfileCellInput;
77
+ }
78
+ /** Campaign result with the same task-failure invariant as `RunRecord`. */
79
+ type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure;
80
+ type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>;
81
+ type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log';
82
+ interface EvalCampaignOptions<V> {
83
+ /**
84
+ * Stable id for the campaign. Used as the default `experimentId` on
85
+ * every run, and folded into the campaign fingerprint.
86
+ */
87
+ campaignId: string;
88
+ variants: CampaignVariant<V>[];
89
+ scenarios: CampaignScenario[];
90
+ /** Default `[0, 1, 2]`. */
91
+ seeds?: number[];
92
+ /** Default `'holdout'` — the split that anchors a launch decision. */
93
+ splitTag?: RunSplitTag;
94
+ /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */
95
+ commitSha: string;
96
+ /**
97
+ * LLM client config. Augmented per-run with `rawSink` and `traceContext`
98
+ * before being passed to the runner. The campaign asserts this config
99
+ * matches `routeRequirements` once at preflight.
100
+ */
101
+ llmOpts: LlmClientOptions;
102
+ /**
103
+ * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail
104
+ * loud if the campaign would silently fall back to the public router or
105
+ * run unauthenticated. Override with an empty object to disable.
106
+ */
107
+ routeRequirements?: LlmRouteRequirements;
108
+ /**
109
+ * Per-run TraceStore factory. Common shape: a fresh store per run keyed
110
+ * on `runId`. Implementations that share a store across the campaign
111
+ * are valid — the campaign only writes through `emitter`.
112
+ */
113
+ storeFactory: (params: CampaignFactoryParams) => TraceStore;
114
+ /**
115
+ * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`
116
+ * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;
117
+ * otherwise required. Forensic capture is non-negotiable in a campaign
118
+ * run — pass `NoopRawProviderSink` explicitly if you want to opt out.
119
+ */
120
+ rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink;
121
+ /**
122
+ * Filesystem root for default `rawSinkFactory`. Ignored if
123
+ * `rawSinkFactory` is supplied.
124
+ */
125
+ workDir?: string;
126
+ /**
127
+ * Extra `onRunComplete` hooks the campaign appends (after its own
128
+ * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.
129
+ */
130
+ onRunComplete?: RunCompleteHook[];
131
+ /**
132
+ * Per-run integrity expectations. Defaults to:
133
+ * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.
134
+ * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.
135
+ */
136
+ integrity?: RunIntegrityExpectations;
137
+ /** Behaviour when integrity fails. Default `'mark_failed'`. */
138
+ onIntegrityFailure?: CampaignIntegrityPolicy;
139
+ /**
140
+ * Per-run runner. Receives a fully-wired context; produces an outcome
141
+ * the campaign converts into a `RunRecord`.
142
+ */
143
+ runner: CampaignRunner<V>;
144
+ /**
145
+ * If set, the campaign computes `researchReport` at the end. `comparator`
146
+ * is a `variantId`. Other fields are forwarded verbatim.
147
+ */
148
+ report?: {
149
+ comparator?: string;
150
+ } & Omit<ResearchReportOptions, 'comparator' | 'preregistrationHash' | 'generatedAt'>;
151
+ /**
152
+ * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).
153
+ * Embedded in the campaign fingerprint and the research report.
154
+ */
155
+ preregistrationHash?: string;
156
+ /** Local concurrency. Default `1` (sequential). */
157
+ concurrency?: number;
158
+ /**
159
+ * Override the time source. Tests pass a mock to make wallMs deterministic.
160
+ */
161
+ now?: () => number;
162
+ /** Override the runId generator. Tests pin this. */
163
+ runId?: (params: CampaignFactoryParams) => string;
164
+ /**
165
+ * Agent profile cell for campaign runs. Static profiles can pass an object;
166
+ * routers or variant-specific harnesses can pass a factory. The campaign
167
+ * stamps the built cell onto every `RunRecord` and rejects profile/model or
168
+ * profile/prompt contradictions.
169
+ */
170
+ agentProfile?: AgentProfileCell | AgentProfileCellInput | ((params: CampaignFactoryParams & {
171
+ variant: V;
172
+ scenarioTags: Record<string, string>;
173
+ }) => AgentProfileCell | AgentProfileCellInput | Promise<AgentProfileCell | AgentProfileCellInput>);
174
+ }
175
+ interface CampaignFactoryParams {
176
+ campaignId: string;
177
+ runId: string;
178
+ variantId: string;
179
+ scenarioId: string;
180
+ seed: number;
181
+ }
182
+ interface FailedRun {
183
+ runId: string;
184
+ variantId: string;
185
+ scenarioId: string;
186
+ seed: number;
187
+ reason: string;
188
+ error?: string;
189
+ }
190
+ interface EvalCampaignResult {
191
+ campaignId: string;
192
+ /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */
193
+ campaignFingerprint: string;
194
+ preregistrationHash: string | null;
195
+ /** Successful runs only. Failed runs land in `failedRuns`. */
196
+ runs: RunRecord[];
197
+ /** Integrity reports for every successful run. */
198
+ integrityReports: RunIntegrityReport[];
199
+ failedRuns: FailedRun[];
200
+ /** Computed when `report` is set on options. */
201
+ report?: ResearchReport;
202
+ startedAt: string;
203
+ endedAt: string;
204
+ }
205
+ declare function runEvalCampaign<V>(opts: EvalCampaignOptions<V>): Promise<EvalCampaignResult>;
206
+ //#endregion
207
+ //#region src/researcher.d.ts
208
+ /** A diagnosed failure mode with the run-IDs that exhibit it. */
209
+ interface FailureMode {
210
+ /** Short machine-readable code. Must be stable across runs of the
211
+ * same researcher to enable longitudinal tracking. */
212
+ code: string;
213
+ /** Human-readable description for the paper / dashboard. */
214
+ description: string;
215
+ evidence: {
216
+ /** Run IDs (from `RunRecord.runId`) where this failure mode was
217
+ * observed. */
218
+ runIds: string[];
219
+ /** Number of run samples that informed the diagnosis. */
220
+ samples: number;
221
+ };
222
+ }
223
+ /** A single steering change the researcher wants to try. */
224
+ interface SteeringChange {
225
+ kind: 'reviewer_prompt' | 'skill_add' | 'skill_remove' | 'threshold' | 'budget';
226
+ /** Implementation-specific payload. Researcher implementations
227
+ * define the schema — keep this `unknown` here to avoid coupling
228
+ * the public interface to any one researcher's internal model. */
229
+ payload: unknown;
230
+ /** Why the researcher proposed this change. Goes into the audit
231
+ * trail next to the failure-mode evidence. */
232
+ rationale: string;
233
+ /** Optional self-reported expected delta on the headline metric. */
234
+ expectedDelta?: number;
235
+ }
236
+ /** A single experiment plan, mapped onto the search/holdout splits. */
237
+ interface ExperimentPlan {
238
+ baselineCandidateId: string;
239
+ proposedCandidateId: string;
240
+ changes: SteeringChange[];
241
+ /** USD ceiling for the entire experiment. The runner must stop
242
+ * before exceeding this and report a partial result. */
243
+ evaluationBudgetUsd: number;
244
+ /** Item IDs (your dataset keys) for the search vs holdout splits. */
245
+ splits: {
246
+ search: string[];
247
+ holdout: string[];
248
+ };
249
+ }
250
+ /** Result of running a plan: every run, plus the gate verdict. */
251
+ interface ExperimentResult {
252
+ plan: ExperimentPlan;
253
+ runs: RunRecord[];
254
+ gateDecision: GateDecision;
255
+ }
256
+ /**
257
+ * The researcher loop. Stable, four-step, inspectable.
258
+ *
259
+ * ┌──────────┐ inspectFailures ┌──────────┐ proposeChange ┌──────────┐
260
+ * │ runs │ ─────────────────▶│ failures │ ──────────────▶│ changes │
261
+ * └──────────┘ └──────────┘ └────┬─────┘
262
+ * │
263
+ * ▼
264
+ * ┌────────────────┐ applyChange ┌────────┐
265
+ * │ ExperimentPlan │ ◀────────────│ base │
266
+ * └────────┬───────┘ └────────┘
267
+ * │
268
+ * evaluateChange ▼
269
+ * ┌────────────────┐
270
+ * │ ExperimentResult│
271
+ * └────────────────┘
272
+ */
273
+ interface Researcher {
274
+ inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
275
+ proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
276
+ applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
277
+ evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
278
+ }
279
+ interface CallbackResearcherOptions {
280
+ inspectFailures: Researcher['inspectFailures'];
281
+ proposeChange: Researcher['proposeChange'];
282
+ applyChange: Researcher['applyChange'];
283
+ evaluateChange: Researcher['evaluateChange'];
284
+ }
285
+ /**
286
+ * Minimal concrete researcher for tests, scripts, and small integrations.
287
+ * Larger autonomous researchers can still implement `Researcher` directly.
288
+ */
289
+ declare class CallbackResearcher implements Researcher {
290
+ private readonly callbacks;
291
+ constructor(callbacks: CallbackResearcherOptions);
292
+ inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
293
+ proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
294
+ applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
295
+ evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
296
+ }
297
+ /**
298
+ * No-op researcher — fails loud on every method. Use as a placeholder
299
+ * in code paths that wire the interface but don't have an implementation
300
+ * yet. Importantly, this does NOT silently succeed: a no-op researcher
301
+ * that returned empty arrays would muffle the loop's signal that
302
+ * nobody implemented the brain.
303
+ */
304
+ declare class NoopResearcher implements Researcher {
305
+ private readonly hint;
306
+ constructor(hint?: string);
307
+ inspectFailures(_runs: RunRecord[]): Promise<FailureMode[]>;
308
+ proposeChange(_failures: FailureMode[]): Promise<SteeringChange[]>;
309
+ applyChange(_changes: SteeringChange[], _baseline: ExperimentPlan): Promise<ExperimentPlan>;
310
+ evaluateChange(_plan: ExperimentPlan): Promise<ExperimentResult>;
311
+ }
312
+ //#endregion
313
+ export { EvalCampaignResult as _, FailureMode as a, SteeringChange as c, CampaignRunContext as d, CampaignRunOutcome as f, EvalCampaignOptions as g, CampaignVariant as h, ExperimentResult as i, CampaignFactoryParams as l, CampaignScenario as m, CallbackResearcherOptions as n, NoopResearcher as o, CampaignRunner as p, ExperimentPlan as r, Researcher as s, CallbackResearcher as t, CampaignIntegrityPolicy as u, FailedRun as v, runEvalCampaign as y };
314
+ //# sourceMappingURL=researcher-CwTdwXG1.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"researcher-CwTdwXG1.d.ts","names":[],"sources":["../src/eval-campaign.ts","../src/researcher.ts"],"mappings":";;;;;;;;UAyEiB,gBAAgB;EAC/B;EACA,SAAS;;UAGM;EACf;;EAEA,OAAO;;UAGQ,mBAAmB;;EAElC;;EAEA;EACA,SAAS;EACT;EACA;EACA,cAAc;EACd;EACA,UAAU;;;;;;;EAOV,SAAS;EACT,OAAO;EACP,SAAS;;;;;;EAMT,SAAS;;UAGD;;EAER;;EAEA;;EAEA;;EAEA,gBAAgB;EAChB,YAAY;;EAEZ;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,gBAAgB;;;;;;EAMhB,cAAc;;;;;;EAMd,eAAe,mBAAmB;;;KAIxB,qBAAqB,2BAA2B;KAEhD,eAAe,MAAM,KAAK,mBAAmB,OAAO,QAAQ;KAE5D;UAEK,oBAAoB;;;;;EAKnC;EACA,UAAU,gBAAgB;EAC1B,WAAW;;EAEX;;EAEA,WAAW;;EAEX;;;;;;EAMA,SAAS;;;;;;EAMT,oBAAoB;;;;;;EAMpB,eAAe,QAAQ,0BAA0B;;;;;;;EAOjD,kBAAkB,QAAQ,0BAA0B;;;;;EAKpD;;;;;EAKA,gBAAgB;;;;;;EAMhB,YAAY;;EAEZ,qBAAqB;;;;;EAKrB,QAAQ,eAAe;;;;;EAKvB;IAAW;MAAwB,KACjC;;;;;EAOF;;EAEA;;;;EAIA;;EAEA,SAAS,QAAQ;;;;;;;EAOjB,eACI,mBACA,0BAEE,QAAQ;IACN,SAAS;IACT,cAAc;QAGd,mBACA,wBACA,QAAQ,mBAAmB;;UAGpB;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;EAEA,MAAM;;EAEN,kBAAkB;EAClB,YAAY;;EAEZ,SAAS;EACT;EACA;;iBAgBoB,gBAAgB,GACpC,MAAM,oBAAoB,KACzB,QAAQ;;;;UC5QM;;;EAGf;;EAEA;EACA;;;IAGE;;IAEA;;;;UAKa;EACf;;;;EAIA;;;EAGA;;EAEA;;;UAIe;EACf;EACA;EACA,SAAS;;;EAGT;;EAEA;IAAU;IAAkB;;;;UAIb;EACf,MAAM;EACN,MAAM;EACN,cAAc;;;;;;;;;;;;;;;;;;;UAoBC;EACf,gBAAgB,MAAM,cAAc,QAAQ;EAC5C,cAAc,UAAU,gBAAgB,QAAQ;EAChD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAC1E,eAAe,MAAM,iBAAiB,QAAQ;;UAG/B;EACf,iBAAiB;EACjB,eAAe;EACf,aAAa;EACb,gBAAgB;;;;;;cAOL,8BAA8B;mBACZ;EAA7B,YAA6B,WAAW;EAExC,gBAAgB,MAAM,cAAc,QAAQ;EAI5C,cAAc,UAAU,gBAAgB,QAAQ;EAIhD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAI1E,eAAe,MAAM,iBAAiB,QAAQ;;;;;;;;;cAYnC,0BAA0B;mBACpB;EAEjB,YAAY;EAIN,gBAAgB,OAAO,cAAc,QAAQ;EAI7C,cAAc,WAAW,gBAAgB,QAAQ;EAIjD,YACJ,UAAU,kBACV,WAAW,iBACV,QAAQ;EAIL,eAAe,OAAO,iBAAiB,QAAQ"}
@@ -0,0 +1,224 @@
1
+ import { s as VerificationReport } from "./multi-layer-verifier-BHY1gWAc.js";
2
+ import { a as RunRecord } from "./run-record-CnZu_gjl.js";
3
+ //#region src/rl/verifiable-reward.d.ts
4
+ type VerifiableRewardSource = 'compile' | 'test' | 'schema' | 'sandbox' | 'judge' | 'composite';
5
+ interface VerifiableReward {
6
+ /** Scalar in [0, 1]. The RL training signal. */
7
+ value: number;
8
+ /** What produced the reward — different sources have different determinism. */
9
+ source: VerifiableRewardSource;
10
+ /**
11
+ * Determinism class. `'deterministic'` rewards are repeatable byte-for-byte
12
+ * given the same inputs (compile, test, schema validation, sandbox exit code).
13
+ * `'probabilistic'` rewards depend on a stochastic component (LLM judge).
14
+ * Mixing these in the same training batch without separation is a known
15
+ * footgun in production RLHF pipelines.
16
+ */
17
+ determinism: 'deterministic' | 'probabilistic';
18
+ /**
19
+ * Confidence in the reward value. For deterministic sources this is 1.0
20
+ * (the bit either flipped or didn't). For judge sources this is the
21
+ * judge-reported confidence or — when missing — a calibrated prior.
22
+ */
23
+ confidence: number;
24
+ /** The layer / judge id that produced the signal, for provenance. */
25
+ origin: string;
26
+ /**
27
+ * Per-source contribution to `value`, keyed by layer/judge id. Single-source
28
+ * rewards carry one entry (`{ [origin]: value }`); composite rewards carry
29
+ * every contributing layer's score — the anti-scalar-collapse surface RL
30
+ * consumers weight per-source instead of trusting one blended number.
31
+ */
32
+ components: Record<string, number>;
33
+ /**
34
+ * The run carries `outcome.realness.gated` — the authenticity gate flagged
35
+ * its success signal as faked.
36
+ *
37
+ * With the gate applied (the default) `value` and every `components` entry
38
+ * are 0 on such a run; with `applyRealnessGate: false` the observed numbers
39
+ * come back untouched and this flag is the only marker that they are not to
40
+ * be trusted. Either way it distinguishes "measured a genuine failure" from
41
+ * "claimed a success we refuse to believe", which a bare 0 cannot.
42
+ */
43
+ realnessGated?: boolean;
44
+ /**
45
+ * Whether an authenticity screen COULD run on this reward at all — the same
46
+ * distinction `RolloutOutcome.realness_screened` draws, for the same reason.
47
+ *
48
+ * `false` on every reward from `extractVerifiableReward`, because a
49
+ * `VerificationReport` carries layer scores and nothing else: there is no
50
+ * `outcome.realness` to consult, so no gate has run, and `realnessGated`
51
+ * being absent there means "unknown", NOT "clean". Absent on the
52
+ * `RunRecord` path when the record itself carries no realness verdict.
53
+ *
54
+ * This matters most exactly where it is easiest to miss: a report whose
55
+ * deterministic layers all passed yields `determinism: 'deterministic'`,
56
+ * `confidence: 1` — the highest-credibility reward this module can emit —
57
+ * and a stubbed integration reporting green is precisely what a gamed run
58
+ * looks like. Consumers driving training off this shape must screen the run
59
+ * themselves; the flag is what tells them nobody has.
60
+ */
61
+ realnessScreened?: boolean;
62
+ }
63
+ interface VerifiableRewardExtractionOptions {
64
+ /**
65
+ * Which layers count as deterministic-reward sources. The verifier doesn't
66
+ * tag layers as "this is verifiable"; the caller declares it via this list
67
+ * (or via the layer name → source mapping). Default treats common names
68
+ * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
69
+ * `sandbox`) as deterministic.
70
+ */
71
+ deterministicLayers?: string[];
72
+ /**
73
+ * Map layer name → reward source. Defaults to a sensible string-match.
74
+ */
75
+ sourceFor?: (layerName: string) => VerifiableRewardSource;
76
+ /**
77
+ * Whether to fall back to a probabilistic (judge) reward when no
78
+ * deterministic layer produced a numeric score. Default `true`. Set to
79
+ * `false` for "deterministic-only" training pipelines that should
80
+ * discard runs without a verifiable signal.
81
+ */
82
+ fallbackToJudge?: boolean;
83
+ /**
84
+ * Default confidence for probabilistic (judge) rewards when the judge
85
+ * doesn't report one. Default `0.7`.
86
+ */
87
+ judgeConfidenceFloor?: number;
88
+ /**
89
+ * Whether the anti-Goodhart realness gate applies. Default `true`, and the
90
+ * default is the one every training path must keep.
91
+ *
92
+ * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
93
+ * for the same reason it reads `observedScore` for its proxy: it measures the
94
+ * DIVERGENCE between the judge signal and the deterministic one, and a
95
+ * deterministic reward that another gate already forced to 0 manufactures
96
+ * exactly that divergence on exactly the gamed population. The detector would
97
+ * then be re-reporting a verdict it was supposed to reach independently.
98
+ */
99
+ applyRealnessGate?: boolean;
100
+ }
101
+ /**
102
+ * Extract a `VerifiableReward` from a `VerificationReport`.
103
+ *
104
+ * Strategy: prefer the deterministic layers (in order: test → compile →
105
+ * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
106
+ * true, return `null` if no signal qualifies. When multiple deterministic
107
+ * layers contribute, return a `'composite'` source with a weighted blend.
108
+ *
109
+ * NO realness gate is applied and none can be: a `VerificationReport` carries
110
+ * layer scores and nothing about whether the run faked them — `realness` lives
111
+ * on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything
112
+ * that becomes training data; this signature is for scoring a report in hand.
113
+ */
114
+ declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
115
+ /**
116
+ * Extract verifiable rewards from `RunRecord[]` produced via the
117
+ * `verificationReportToRunRecord` adapter (which encodes per-layer scores
118
+ * in `outcome.raw['layer.<name>']`). For records that don't carry layer
119
+ * scores, returns `null` for that record.
120
+ *
121
+ * This is the canonical bridge from "campaign-shaped artifacts" to
122
+ * "RL-training-ready reward signals": every record that has a clean
123
+ * verifiable reward becomes a training datum, every record that doesn't
124
+ * gets filtered out (or kept with `'probabilistic'` determinism for
125
+ * separate downstream handling).
126
+ *
127
+ * The realness gate applies to EVERY channel here, and to the deterministic one
128
+ * MOST. It is tempting to reason that a decidable signal cannot be gamed, so
129
+ * the gate is redundant on it — that reasoning is backwards. `realness.gated`
130
+ * means the run's success signal was FAKED, and a test suite reporting green on
131
+ * a stubbed integration is precisely what that looks like: the deterministic
132
+ * layer is the thing that got faked. Exporting it ungated hands a trainer the
133
+ * highest-credibility reward the module can emit (`determinism: 'deterministic'`,
134
+ * `confidence: 1`) for the one population the gate exists to catch. Pass
135
+ * `applyRealnessGate: false` only to look at the ungated numbers for detection.
136
+ */
137
+ declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
138
+ runId: string;
139
+ reward: VerifiableReward | null;
140
+ }>;
141
+ /**
142
+ * Filter `RunRecord[]` to those with deterministic verifiable rewards.
143
+ *
144
+ * A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the
145
+ * same rule GRPO uses on a gated line. 0 is the honest label for a faked
146
+ * success and is usable signal, whereas dropping the run would move a group
147
+ * baseline without saying so. (SFT differs: there every row is a target to
148
+ * imitate, so a gated row is removed outright.)
149
+ */
150
+ declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
151
+ run: RunRecord;
152
+ reward: VerifiableReward;
153
+ }>;
154
+ //#endregion
155
+ //#region src/rl/reward-hacking.d.ts
156
+ type RewardHackingSignal = 'reward_divergence' | 'distribution_shift' | 'reward_disagreement' | 'judge_drift';
157
+ interface RewardHackingFinding {
158
+ signal: RewardHackingSignal;
159
+ /** Severity in [0, 1]. >0.5 = strong signal. */
160
+ severity: number;
161
+ message: string;
162
+ /** Numeric evidence the consumer can render. */
163
+ detail: Record<string, number>;
164
+ }
165
+ interface RewardHackingReport {
166
+ findings: RewardHackingFinding[];
167
+ /** Signals with enough usable observations to produce a finding. */
168
+ evaluatedSignals: RewardHackingSignal[];
169
+ /**
170
+ * Composite verdict. `'insufficient_evidence'` when fewer than four scored
171
+ * runs exist; otherwise `'clean'` if every signal severity < 0.3,
172
+ * `'suspect'` if at least one ≥ 0.3 but none ≥ 0.6, and `'gaming'` if any ≥ 0.6.
173
+ */
174
+ verdict: 'insufficient_evidence' | 'clean' | 'suspect' | 'gaming';
175
+ /** Rationale for the verdict, ready to paste into an audit log. */
176
+ rationale: string[];
177
+ /** Number of runs with a usable proxy reward. */
178
+ n: number;
179
+ }
180
+ interface DetectRewardHackingInput {
181
+ /**
182
+ * Run records ordered by recency (oldest first). The detector segments
183
+ * them into prefix/suffix windows to compute "did the gap widen."
184
+ */
185
+ runs: RunRecord[];
186
+ /**
187
+ * The metric the policy was trained to optimize. Should be present on
188
+ * `outcome.raw` or `outcome.holdoutScore`. Default reads `outcome.holdoutScore`.
189
+ */
190
+ proxyOf?: (run: RunRecord) => number | null;
191
+ /**
192
+ * The held-out ground-truth metric. For RL on coding, this is typically
193
+ * test pass-rate. For RLHF, it's downstream task performance or human
194
+ * preference. For knowledge tasks, it's an independently-graded score.
195
+ */
196
+ truthOf?: (run: RunRecord) => number | null;
197
+ /**
198
+ * Independent secondary reward. Used for the `reward_disagreement`
199
+ * signal. Default uses the verifiable reward extractor (deterministic
200
+ * sources only).
201
+ */
202
+ secondaryRewardOf?: (run: RunRecord) => number | null;
203
+ /**
204
+ * Window size — how many of the most recent runs count as the "after"
205
+ * cohort. Default min(50, half the runs).
206
+ */
207
+ windowSize?: number;
208
+ /**
209
+ * Severity threshold to flag a signal. Default 0.3 (suspect) and 0.6
210
+ * (gaming).
211
+ */
212
+ thresholds?: {
213
+ suspect?: number;
214
+ gaming?: number;
215
+ };
216
+ /**
217
+ * Verifiable-reward options used for the secondary-reward fallback.
218
+ */
219
+ verifiableRewardOptions?: VerifiableRewardExtractionOptions;
220
+ }
221
+ declare function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport;
222
+ //#endregion
223
+ export { detectRewardHacking as a, VerifiableRewardSource as c, filterDeterministicallyRewarded as d, RewardHackingSignal as i, extractVerifiableReward as l, RewardHackingFinding as n, VerifiableReward as o, RewardHackingReport as r, VerifiableRewardExtractionOptions as s, DetectRewardHackingInput as t, extractVerifiableRewardsFromRecords as u };
224
+ //# sourceMappingURL=reward-hacking-eAnOsynk.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"reward-hacking-eAnOsynk.d.ts","names":[],"sources":["../src/rl/verifiable-reward.ts","../src/rl/reward-hacking.ts"],"mappings":";;;KAiCY;UAQK;;EAEf;;EAEA,QAAQ;;;;;;;;EAQR;;;;;;EAMA;;EAEA;;;;;;;EAOA,YAAY;;;;;;;;;;;EAWZ;;;;;;;;;;;;;;;;;;EAkBA;;UAGe;;;;;;;;EAQf;;;;EAIA,aAAa,sBAAsB;;;;;;;EAOnC;;;;;EAKA;;;;;;;;;;;;EAYA;;;;;;;;;;;;;;;iBA6Cc,wBACd,QAAQ,oBACR,OAAM,oCACL;;;;;;;;;;;;;;;;;;;;;;;iBAqGa,oCACd,MAAM,aACN,OAAM,oCACL;EAAQ;EAAe,QAAQ;;;;;;;;;;;iBA4GlB,gCACd,MAAM,aACN,OAAM,oCACL;EAAQ,KAAK;EAAW,QAAQ;;;;KChWvB;UAMK;EACf,QAAQ;;EAER;EACA;;EAEA,QAAQ;;UAGO;EACf,UAAU;;EAEV,kBAAkB;;;;;;EAMlB;;EAEA;;EAEA;;UAGe;;;;;EAKf,MAAM;;;;;EAKN,WAAW,KAAK;;;;;;EAMhB,WAAW,KAAK;;;;;;EAMhB,qBAAqB,KAAK;;;;;EAK1B;;;;;EAKA;IAAe;IAAkB;;;;;EAIjC,0BAA0B;;iBAgBZ,oBAAoB,OAAO,2BAA2B"}