@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
@@ -1,1312 +1,6 @@
1
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
2
- type AgentProfileDimensionValue = string | number | boolean | null;
3
- interface AgentProfileSource {
4
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
5
- kind: string;
6
- /** sha256 over the canonical source profile object. */
7
- hash: string;
8
- }
9
- interface AgentProfileHarness {
10
- id: string;
11
- version?: string;
12
- hash?: string;
13
- }
14
- interface AgentProfileCell {
15
- schemaVersion: AgentProfileCellSchemaVersion;
16
- cellId: string;
17
- profileId: string;
18
- sourceProfile: AgentProfileSource;
19
- harness?: AgentProfileHarness;
20
- model?: string;
21
- promptHash?: string;
22
- dimensions?: Record<string, AgentProfileDimensionValue>;
23
- }
24
-
25
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
26
-
27
- /**
28
- * Paper-grade RunRecord schema + runtime validator.
29
- *
30
- * Every run that participates in a promotion gate, paper table, or
31
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
32
- * fields are exactly those the paper "Two Loops, Three Roles" requires
33
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
34
- * holdout split tag. A task score is optional because execution-only records
35
- * must preserve missing labels instead of converting errors into zero quality.
36
- *
37
- * This is intentionally NOT a replacement for the rich `Run` /
38
- * `ProposeReviewReport` / `ScenarioResult` types already in the
39
- * package. Those are runtime structures with full provenance. A
40
- * `RunRecord` is the analysis-time projection — the JSON-friendly
41
- * row you'd put in a parquet file or paste into a notebook.
42
- *
43
- * Validate at the boundary:
44
- *
45
- * const rec = validateRunRecord(rawJson) // throws on missing
46
- * const ok = isRunRecord(rawJson) // boolean check
47
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
48
- *
49
- * The validator runs in pure TS — zod is intentionally NOT a
50
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
51
- */
52
-
53
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
54
- * combined train+test pool that the optimizer is allowed to read. */
55
- type RunSplitTag = 'search' | 'dev' | 'holdout';
56
- /**
57
- * Explicit execution-lifecycle result for a run.
58
- *
59
- * This is separate from task quality (`outcome`) and failure classification.
60
- * Producers set it only from root-run or process evidence.
61
- */
62
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
63
- interface RunTokenUsage {
64
- input: number;
65
- /** All generated tokens charged as output, including reasoning tokens. */
66
- output: number;
67
- /** Reasoning-token subset of `output`, when the provider reports it. */
68
- reasoning?: number;
69
- /** Prompt tokens served from a provider cache. */
70
- cached?: number;
71
- /** Prompt tokens written into a provider cache. */
72
- cacheWrite?: number;
73
- }
74
- /**
75
- * How a run's USD amount was obtained.
76
- */
77
- type RunCostProvenance = {
78
- kind: 'observed';
79
- usd: number;
80
- } | {
81
- kind: 'estimated';
82
- usd: number;
83
- } | {
84
- kind: 'uncaptured';
85
- usd: null;
86
- };
87
- interface RunJudgeMetadata {
88
- model: string;
89
- promptVersion: string;
90
- /** [0,1] confidence the judge declared. Constant judge confidence
91
- * across many runs is a fallback signal (see `canary.ts`). */
92
- confidence: number;
93
- /** True if the judge degraded to a fallback path (rules-only,
94
- * prior-call cache, etc.). The canary uses this to alert. */
95
- fallback: boolean;
96
- }
97
- /**
98
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
99
- * judges over a multi-dimensional rubric.
100
- *
101
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
102
- * composite the gate uses. The full breakdown belongs here so consumers
103
- * can answer "which judge disagreed?", "which dimension dragged the
104
- * composite down?", and "did half the panel fail?" without re-running.
105
- *
106
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
107
- * `composite` are convenience projections — derivable but precomputed so
108
- * downstream IRR primitives (`interRaterReliability`,
109
- * `corpusInterRaterAgreement`) and reporters don't pay the same
110
- * aggregation twice.
111
- *
112
- * Fail-loud discipline: judges that errored out land in `failedJudges`
113
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
114
- * run); the explicit list makes a partial-failure recorded as such.
115
- */
116
- interface JudgeScoresRecord {
117
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
118
- perJudge: Record<string, Record<string, number>>;
119
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
120
- perDimMean: Record<string, number>;
121
- /** Composite mean across successful judges. Mirrors the task score only
122
- * when `failedJudges` is empty. */
123
- composite: number;
124
- /** Judges that errored or returned an unparseable verdict. Recorded
125
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
126
- * not inferred from missing keys in `perJudge`. */
127
- failedJudges?: string[];
128
- /** Free-form notes the judges emitted (joined across judges or
129
- * first-judge only — consumer's choice). */
130
- notes?: string;
131
- }
132
- interface RunOutcome {
133
- /** Score on the search/optimization split. Optional for holdout-only and
134
- * execution-only records. */
135
- searchScore?: number;
136
- /** Score on the held-out split. Optional for search-only and execution-only
137
- * records. When both scores are absent, the run is explicitly unlabeled. */
138
- holdoutScore?: number;
139
- /** Bag of any other metric the run produced — judge dimensions,
140
- * pass/fail counters, latency stats, etc. Numeric only — keeps
141
- * reporters honest. */
142
- raw: Record<string, number>;
143
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
144
- * judgements populate this; substrate primitives like
145
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
146
- * these records as input. Optional — single-judge or scalar-only
147
- * runs leave it unset. */
148
- judgeScores?: JudgeScoresRecord;
149
- /** Authenticity / realness verdict — did the run build the REAL thing on the
150
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
151
- * with an authenticity config populate it. Carried in the corpus so the
152
- * flywheel / off-policy learning can optimize for real completion, not gamed
153
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
154
- * must not count as a real success regardless of `score`. */
155
- realness?: {
156
- score: number;
157
- gated: boolean;
158
- reason?: string;
159
- };
160
- }
161
- /**
162
- * Mandatory paper-grade fields for a single evaluation run. Optional
163
- * fields are extension points; mandatory fields throw if missing.
164
- *
165
- * Hash discipline:
166
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
167
- * model (after any steering bundle merge).
168
- * - `configHash` is the sha256 of the effective run config (model,
169
- * temperature, tools, judges, splits). The pair (promptHash,
170
- * configHash) uniquely identifies an experiment cell.
171
- *
172
- * Model snapshot discipline:
173
- * - `model` MUST encode a snapshot version. Bare aliases like
174
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
175
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
176
- */
177
- interface RunRecord {
178
- /** UUID for the run. */
179
- runId: string;
180
- /** Logical experiment grouping (a treatment vs a baseline within
181
- * the same sweep should share `experimentId`). */
182
- experimentId: string;
183
- /** Stable identifier for the candidate (variant) being run. The
184
- * promotion gate compares two `candidateId`s on matched items. */
185
- candidateId: string;
186
- /** RNG seed for the run. Always recorded — silent re-seeding is
187
- * the most common cause of non-reproducible numbers. */
188
- seed: number;
189
- /** Model identifier WITH snapshot version. */
190
- model: string;
191
- /** sha256 of the effective prompt (post-steering). */
192
- promptHash: string;
193
- /** sha256 of the effective config. */
194
- configHash: string;
195
- /** Git SHA the harness was run from. */
196
- commitSha: string;
197
- /** End-to-end wall-clock duration in milliseconds. */
198
- wallMs: number;
199
- /** Time spent queued before execution started, if known. */
200
- queueMs?: number;
201
- /** Total USD cost, or null when the producer could not capture one. */
202
- costUsd: number | null;
203
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
204
- costProvenance: RunCostProvenance;
205
- /** Token usage breakdown. */
206
- tokenUsage: RunTokenUsage;
207
- /** Root-run or process terminal result. Never inferred from a child span. */
208
- terminalOutcome: RunTerminalOutcome;
209
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
210
- * or incomplete terminal result; never populated from a child span. */
211
- terminalFailureReason?: string;
212
- /** Judge-side metadata, if a judge was used. */
213
- judgeMetadata?: RunJudgeMetadata;
214
- /** Per-split scores + raw bag. */
215
- outcome: RunOutcome;
216
- /** Canonical task-failure class drawn from the shared
217
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
218
- * evidence. Execution errors belong in
219
- * `outcome.raw.execution_error_count`. */
220
- failureClass?: FailureClass;
221
- /** Free-form task-failure detail scoped under a non-success
222
- * `failureClass`. It is invalid without that class. */
223
- failureMode?: string;
224
- /** Which split this run was drawn from. */
225
- splitTag: RunSplitTag;
226
- /**
227
- * Stable scenario identifier the run observed or was scored against.
228
- * Comparison primitives match this identity rather than input order.
229
- */
230
- scenarioId: string;
231
- /**
232
- * Canonical identity for the agent profile cell that produced this row:
233
- * profile artifact hash plus optional harness/model/prompt/reporting
234
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
235
- * longitudinal reports by the complete source profile, not by a loose
236
- * candidate label or opaque config hash.
237
- */
238
- agentProfile?: AgentProfileCell;
239
- }
240
-
241
- /**
242
- * OutcomeStore — deployment outcomes attached to Run IDs.
243
- *
244
- * Outcomes arrive asynchronously from production telemetry after the
245
- * eval run completed: user ratings, retention flags, conversion events,
246
- * revenue, support-ticket rate, anything a product team can measure.
247
- * The store is a peer to TraceStore — separate lifecycle, same runId
248
- * foreign key.
249
- *
250
- * The whole point of this module is to make the meta-eval correlation
251
- * question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.
252
- */
253
- interface DeploymentOutcome {
254
- runId: string;
255
- capturedAt: number;
256
- /** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */
257
- metrics: Record<string, number>;
258
- /** Dimensions for stratified analysis — cohort, region, user_segment. */
259
- labels?: Record<string, string>;
260
- /** Free-form provenance (source system, pipeline version). */
261
- source?: string;
262
- }
263
- interface OutcomeFilter {
264
- runIds?: string[];
265
- since?: number;
266
- until?: number;
267
- label?: {
268
- key: string;
269
- value: string;
270
- };
271
- source?: string;
272
- }
273
- interface OutcomeStore {
274
- append(outcome: DeploymentOutcome): Promise<void>;
275
- /** All outcomes attached to this run (a single run can have many — multiple
276
- * capture windows over deployment time). */
277
- forRun(runId: string): Promise<DeploymentOutcome[]>;
278
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
279
- }
280
-
281
- /**
282
- * Rubric predictive validity — does our eval rubric predict deployment
283
- * outcomes?
284
- *
285
- * `correlationStudy` (already in this package) joins a `TraceStore` to an
286
- * `OutcomeStore` and computes Pearson + Spearman + bootstrap CI for each
287
- * (eval-metric, outcome-metric) pair. That answers "does X correlate with
288
- * Y at all." `rubricPredictiveValidity` is the campaign-shaped wrapper
289
- * around it: take a sequence of `RunRecord`s (the canonical campaign
290
- * artifact) and a `DeploymentOutcomeStore`, join on `runId`, return a
291
- * ranked verdict on every rubric whose dimension scores were captured in
292
- * `outcome.raw`.
293
- *
294
- * The point — quoting the methodology doc — is that **without this loop
295
- * every rubric is faith-based**. Once it's wired, you know which rubrics
296
- * have earned their promotion power and which ones are decoration.
297
- *
298
- * const validity = await rubricPredictiveValidity({
299
- * runs: lastQuarter,
300
- * outcomes: shipFlagOutcomeStore,
301
- * outcomeMetrics: ['revenue_lift', 'retention_30d', 'csat'],
302
- * rubrics: ['anti_slop', 'semantic_concept', 'tool_recovery'],
303
- * })
304
- * for (const r of validity.ranked) {
305
- * console.log(`${r.rubric} → ${r.bestOutcome}: ρ=${r.spearman.toFixed(2)}`)
306
- * }
307
- *
308
- * The function is intentionally read-only. Use the verdict to deprecate
309
- * decorative rubrics, re-weight composite scores, or trigger a
310
- * recalibration sweep when predictive validity drops below a threshold.
311
- */
312
-
313
- interface RubricPredictiveValidityInput {
314
- /**
315
- * Canonical campaign output. Each record's `outcome.raw[<rubricId>]`
316
- * provides the eval score; missing keys are silently skipped per pair.
317
- */
318
- runs: RunRecord[];
319
- outcomes: OutcomeStore;
320
- /**
321
- * Outcome metric names to evaluate against. Each must appear in at
322
- * least one `DeploymentOutcome.metrics` keyspace; pairs with too few
323
- * joined samples are excluded from the result.
324
- */
325
- outcomeMetrics: string[];
326
- /**
327
- * Rubric ids to evaluate. Must appear as keys in `RunRecord.outcome.raw`.
328
- * If omitted, every numeric key in `outcome.raw` across the run set is
329
- * treated as a rubric.
330
- */
331
- rubrics?: string[];
332
- /** Minimum joined-sample count before a pair is reported. Default 8. */
333
- minSamples?: number;
334
- /** Bootstrap resamples for CI. Default 500. */
335
- bootstrapResamples?: number;
336
- /** Random seed for the bootstrap (mulberry32). Default unset (Math.random). */
337
- seed?: number;
338
- /**
339
- * Reduction when multiple outcomes attach to one runId. Default `'latest'`
340
- * (most recently captured).
341
- */
342
- reduction?: 'latest' | 'mean' | 'max';
343
- }
344
- interface RubricOutcomePair {
345
- rubric: string;
346
- outcome: string;
347
- n: number;
348
- pearson: number;
349
- spearman: number;
350
- ci95: {
351
- low: number;
352
- high: number;
353
- };
354
- /**
355
- * Verdict bucket. `load_bearing` ≥ 0.7, `informative` ≥ 0.4,
356
- * `decorative` < 0.4 in absolute correlation. A negative correlation
357
- * with a desired outcome is also `decorative` — actively misleading
358
- * is worse than uninformative.
359
- */
360
- verdict: 'load_bearing' | 'informative' | 'decorative';
361
- }
362
- interface RubricRanking {
363
- rubric: string;
364
- /** Outcome metric this rubric correlated best with. */
365
- bestOutcome: string;
366
- spearman: number;
367
- pearson: number;
368
- n: number;
369
- verdict: RubricOutcomePair['verdict'];
370
- }
371
- interface RubricPredictiveValidityReport {
372
- pairs: RubricOutcomePair[];
373
- /** Per-rubric best pair, sorted descending by |spearman|. */
374
- ranked: RubricRanking[];
375
- joinedSamples: number;
376
- skippedRuns: number;
377
- /** Rubrics that were declared but never produced a usable score. */
378
- rubricsWithoutData: string[];
379
- }
380
- declare function rubricPredictiveValidity(input: RubricPredictiveValidityInput): Promise<RubricPredictiveValidityReport>;
381
-
382
- /**
383
- * Bootstrap-CI promotion gate.
384
- *
385
- * In any iterative-improvement loop (GEPA, prompt evolution, dataset
386
- * curation), the question is "did this generation actually improve, or are
387
- * we celebrating noise?". With small N and noisy outcomes, point-estimate
388
- * deltas lie. Bootstrap confidence intervals tell the operator whether the
389
- * delta is real before code or prompts get promoted.
390
- *
391
- * This module is pure functions — no I/O, no model calls. Easy to unit-test
392
- * and to compose into any verdict gate.
393
- *
394
- * Default gate:
395
- * - Bootstrap mean baseline vs candidate (1k resamples).
396
- * - Compute the delta distribution; pass if the lower CI bound > 0.
397
- * - Tunable confidence (default 95%) and resample count.
398
- *
399
- * Verdict semantics intentionally match the existing `experiments.jsonl`
400
- * vocabulary:
401
- * - ADVANCE: candidate's CI lower bound > baseline mean (real win)
402
- * - KEEP: overlap, but candidate point estimate >= baseline (neutral)
403
- * - REVERT: candidate's CI upper bound < baseline mean (real regression)
404
- * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal
405
- */
406
- type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE';
407
- interface BootstrapResult {
408
- baselineMean: number;
409
- candidateMean: number;
410
- /** candidateMean - baselineMean, point estimate. */
411
- delta: number;
412
- /** Lower bound of the (1 - alpha) CI on the delta. */
413
- ciLower: number;
414
- /** Upper bound of the (1 - alpha) CI on the delta. */
415
- ciUpper: number;
416
- /** Number of bootstrap resamples used. */
417
- iterations: number;
418
- alpha: number;
419
- verdict: Verdict;
420
- }
421
- interface BootstrapOptions {
422
- /** Confidence level alpha (default 0.05 → 95% CI). */
423
- alpha?: number;
424
- /** Number of resamples (default 1000). */
425
- iterations?: number;
426
- /**
427
- * Minimum total samples (baseline + candidate) below which we always
428
- * return INCONCLUSIVE — bootstrap with too few samples is meaningless.
429
- * Default 6 (combined).
430
- */
431
- minTotalSamples?: number;
432
- /** RNG seed for reproducibility. Default: Math.random. */
433
- seed?: number;
434
- }
435
- /**
436
- * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.
437
- *
438
- * Uses simple percentile bootstrap on the difference of resampled means.
439
- * That's the standard non-parametric primitive — no distributional
440
- * assumptions, robust to skew, easy to reason about.
441
- */
442
- declare function bootstrapCi(baseline: number[], candidate: number[], options?: BootstrapOptions): BootstrapResult;
443
- /**
444
- * Judge-replay promotion gate.
445
- *
446
- * The cheap inner-loop judge that drives an evolution run is by definition
447
- * fast and noisy. When you're about to promote a winning variant to the
448
- * canonical default, you want a STRONGER judge (a more expensive model, a
449
- * human grader, a separately-trained reward model) to confirm the win
450
- * generalises beyond the inner loop.
451
- *
452
- * This helper takes raw winner + baseline outputs, scores both through the
453
- * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger
454
- * judge agrees the winner is real with the configured confidence. Doesn't
455
- * matter what shape your "output" is — pass a string, an object, anything
456
- * the judge can read.
457
- */
458
- interface JudgeReplayGateArgs<TOutput> {
459
- baselineOutputs: TOutput[];
460
- candidateOutputs: TOutput[];
461
- /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */
462
- judge: (output: TOutput) => Promise<number> | number;
463
- alpha?: number;
464
- iterations?: number;
465
- /** RNG seed for reproducibility. */
466
- seed?: number;
467
- /** Maximum concurrent judge calls. Default 4. */
468
- judgeConcurrency?: number;
469
- }
470
- /**
471
- * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
472
- */
473
- declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
474
- baselineSamples: number;
475
- candidateSamples: number;
476
- }>;
477
-
478
- /**
479
- * Dataset — versioned, sliceable, content-hashed scenario collection.
480
- *
481
- * Scenarios stop being ephemeral arrays and become first-class
482
- * artifacts. Every Dataset carries:
483
- * - content hash (sha256 over canonicalized scenario array)
484
- * - provenance (contributor, createdAt, sourceUrl)
485
- * - split labels (train | dev | test | holdout)
486
- * - difficulty tiers (easy | medium | hard | extreme)
487
- * - tags (free-form, per-scenario)
488
- *
489
- * `Dataset.slice({ difficulty, split, holdout, seed })` returns a
490
- * deterministic, reproducible subset. Holdout slices are locked: you
491
- * can read them but `mutate` throws, which prevents "oh I'll just
492
- * tweak that one scenario" contamination drift.
493
- */
494
- type DatasetSplit = 'train' | 'dev' | 'test' | 'holdout';
495
- type DatasetDifficulty = 'easy' | 'medium' | 'hard' | 'extreme';
496
- interface DatasetScenario {
497
- id: string;
498
- /** Arbitrary payload; the framework doesn't interpret it. */
499
- payload: unknown;
500
- split?: DatasetSplit;
501
- difficulty?: DatasetDifficulty;
502
- /** Canary token that MUST NOT round-trip through a correct agent output. */
503
- canary?: string;
504
- /**
505
- * Behavioral-canary forbidden pattern. A string OR a serialized regex
506
- * (`/.../flags`) that the agent under test MUST NOT emit. Used by
507
- * {@link import('./canary').checkBehavioralCanary | checkBehavioralCanary},
508
- * which inverts the contamination-style semantic: presence in the
509
- * agent output is a LEAK / failure, not a positive signal.
510
- *
511
- * Falls back to {@link canary} when omitted.
512
- */
513
- forbiddenPattern?: string;
514
- tags?: Record<string, string>;
515
- }
516
- interface DatasetProvenance {
517
- contributor?: string;
518
- createdAt: string;
519
- sourceUrl?: string;
520
- license?: string;
521
- description?: string;
522
- /** Monotonic human-readable version (e.g. "2026.04.20"). */
523
- version: string;
524
- }
525
- interface DatasetManifest {
526
- name: string;
527
- provenance: DatasetProvenance;
528
- /** sha256 hex over canonicalized scenarios. */
529
- contentHash: string;
530
- scenarioCount: number;
531
- splitCounts: Record<DatasetSplit, number>;
532
- }
533
-
534
- /**
535
- * HeldOutGate — first-class held-out paired-delta promotion gate.
536
- *
537
- * Encodes the "honesty override" pattern that lived inline in
538
- * `~/webb/redteam/scripts/agent-eval-autoresearch.ts:138–171`.
539
- * The optimizer's best-guess is one thing; what we should actually
540
- * ship is another. The gate is the line between them.
541
- *
542
- * A candidate is promoted iff ALL three pass:
543
- *
544
- * 1. **Productive runs**: the candidate has at least
545
- * `minProductiveRuns` paired observations on items where BOTH
546
- * candidate and baseline produced a real (non-silent) score.
547
- * 2. **Paired delta**: the lower bound of the bootstrap CI on the
548
- * median per-item delta (candidate − baseline) on the HOLDOUT
549
- * split is strictly greater than `pairedDeltaThreshold`.
550
- * 3. **Overfit gap**: the candidate's gap between search-split
551
- * score and holdout-split score is no worse (more positive)
552
- * than the baseline's gap by more than `overfitGapThreshold`.
553
- * "Better on search, worse on holdout" is the canonical
554
- * overfit pattern; this catches it.
555
- *
556
- * The decision carries a machine-readable `rejectionCode` plus an
557
- * `evidence` block with every number the gate looked at, so the
558
- * downstream researcher / paper / dashboard can re-derive the
559
- * verdict without re-running.
560
- *
561
- * See also:
562
- * - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank`
563
- * - `src/run-record.ts` for the input row schema
564
- * - `src/reference-replay.ts` for the older, reference-replay-
565
- * specific promotion path (still useful for replay-style evals).
566
- */
567
-
568
- type HeldOutGateRejectionCode = 'few_runs' | 'missing_split_scores' | 'missing_cost' | 'negative_delta' | 'overfit_gap' | 'cost_ceiling';
569
- interface GateEvidence {
570
- /** Number of paired (candidate, baseline) holdout observations used. */
571
- productiveRuns: number;
572
- /** Candidate holdout rows with no baseline row at the same work identity. */
573
- unpairedCandidateRuns: number;
574
- /** Baseline holdout rows with no candidate row at the same work identity. */
575
- unpairedBaselineRuns: number;
576
- /** Median of paired holdout deltas, or null when there are no pairs. */
577
- medianPairedDelta: number | null;
578
- /** Bootstrap CI on the median paired holdout delta, if computed. */
579
- pairedCI: {
580
- low: number;
581
- high: number;
582
- } | null;
583
- /** Wilcoxon signed-rank p-value, if computed. */
584
- pairedPValue: number | null;
585
- /** Mean candidate score on the search split, or null when absent. */
586
- searchScore: number | null;
587
- /** Mean candidate score on the holdout split, or null when absent. */
588
- holdoutScore: number | null;
589
- /** Candidate (search − holdout) gap, or null when either side is absent. */
590
- overfitGap: number | null;
591
- /** Baseline (search − holdout) gap, or null when either side is absent. */
592
- baselineOverfitGap: number | null;
593
- /** Median per-task USD cost across the candidate's runs. Recorded
594
- * even when no `costPerTaskCeiling` is configured so downstream
595
- * dashboards (intelligence.tangle.tools) can render \$/task per
596
- * generation regardless of gating policy. */
597
- medianCandidateCost: number | null;
598
- /** Median per-task USD cost across the baseline runs, for
599
- * symmetric reporting. */
600
- medianBaselineCost: number | null;
601
- /**
602
- * Runs (candidate + baseline) dropped before pairing because the
603
- * authenticity gate flagged them as gamed. Surfaced rather than silent: a
604
- * promotion decision computed over a shrunken pool has to say by how much,
605
- * and a nonzero count here is itself the finding.
606
- */
607
- realnessGatedRuns: number;
608
- }
609
- interface GateDecision {
610
- /** Final promote/no-promote verdict. */
611
- promote: boolean;
612
- /** The candidate that was evaluated. */
613
- candidateId: string;
614
- /** The baseline it was compared against. */
615
- baselineId: string;
616
- /** Every number the gate looked at, for audit + paper export. */
617
- evidence: GateEvidence;
618
- /** Human-readable reason. */
619
- reason: string;
620
- /** Machine-readable rejection code, or null on promote. */
621
- rejectionCode: HeldOutGateRejectionCode | null;
622
- }
623
-
624
- /**
625
- * Release confidence gate.
626
- *
627
- * This is the production-facing composition layer over the lower-level
628
- * primitives:
629
- * - Dataset manifests prove corpus/version coverage.
630
- * - RunRecord rows prove reproducible search/holdout outcomes.
631
- * - Multi-shot trace evidence carries turn counts and ASI diagnostics.
632
- * - HeldOutGate decisions remain the paired promotion authority.
633
- *
634
- * The gate is intentionally pure and conservative. Missing declared evidence
635
- * fails closed instead of being treated as a neutral zero.
636
- */
637
-
638
- /** Severity of an actionable finding attached to a run/trace. */
639
- type AsiSeverity = 'info' | 'warning' | 'error' | 'critical';
640
- /** Actionable side-info — a diagnosed finding the loop can act on. */
641
- interface ActionableSideInfo {
642
- /** Stable expectation/check id when available. */
643
- expectationId?: string;
644
- /** Human-readable diagnosis of what happened. */
645
- message: string;
646
- severity?: AsiSeverity;
647
- /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */
648
- evidence?: string;
649
- /** Prompt/tool/context surface likely responsible. */
650
- responsibleSurface?: string;
651
- /** Suggested fix in natural language. */
652
- suggestion?: string;
653
- /** Whether this expectation was satisfied. Defaults to false for ASI rows. */
654
- matched?: boolean;
655
- metadata?: Record<string, unknown>;
656
- }
657
- type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail';
658
- type ReleaseConfidenceAxisName = 'corpus' | 'quality' | 'reliability' | 'generalization' | 'diagnostics' | 'efficiency';
659
- interface ReleaseTraceEvidence {
660
- scenarioId: string;
661
- candidateId?: string;
662
- split?: RunSplitTag;
663
- score?: number;
664
- ok?: boolean;
665
- turnCount?: number;
666
- costUsd?: number;
667
- durationMs?: number;
668
- /** Canonical task-failure class. Free-form detail belongs in ASI. */
669
- failureClass?: FailureClass;
670
- asi?: ActionableSideInfo[];
671
- metadata?: Record<string, unknown>;
672
- }
673
- interface ReleaseConfidenceThresholds {
674
- /** Require a Dataset manifest or explicit scenarios. Default true. */
675
- requireCorpus?: boolean;
676
- minScenarioCount?: number;
677
- minSearchRuns?: number;
678
- minHoldoutRuns?: number;
679
- /** Require at least one holdout scenario/run. Default true. */
680
- requireHoldout?: boolean;
681
- minPassRate?: number;
682
- minMeanScore?: number;
683
- /** Search mean may exceed holdout mean by at most this much. */
684
- maxOverfitGap?: number;
685
- maxMeanCostUsd?: number;
686
- maxP95WallMs?: number;
687
- /** Low-score/failed rows must carry ASI. Default true. */
688
- requireAsiForFailures?: boolean;
689
- /** Score below this is considered a failure for ASI coverage. Default 0.5. */
690
- failureScoreThreshold?: number;
691
- }
692
- interface ReleaseConfidenceInput {
693
- target: string;
694
- candidateId?: string;
695
- baselineId?: string;
696
- dataset?: DatasetManifest;
697
- scenarios?: readonly DatasetScenario[];
698
- runs?: readonly RunRecord[];
699
- traces?: readonly ReleaseTraceEvidence[];
700
- gateDecision?: GateDecision | null;
701
- thresholds?: ReleaseConfidenceThresholds;
702
- }
703
- interface ReleaseConfidenceAxis {
704
- name: ReleaseConfidenceAxisName;
705
- status: ReleaseConfidenceStatus;
706
- score: number | null;
707
- detail: string;
708
- }
709
- interface ReleaseConfidenceIssue {
710
- axis: ReleaseConfidenceAxisName;
711
- severity: 'critical' | 'warning';
712
- code: string;
713
- detail: string;
714
- }
715
- interface ReleaseConfidenceMetrics {
716
- scenarioCount: number;
717
- /** Search rows with a finite search score. */
718
- searchRuns: number;
719
- /** Holdout rows with a finite holdout score. */
720
- holdoutRuns: number;
721
- /** Runs with neither a split-matched score nor an explicit task failure. */
722
- unscoredRuns: number;
723
- /** Run rows, or trace rows when no runs exist, with no classified terminal result. */
724
- unclassifiedTerminalRuns: number;
725
- /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */
726
- terminalFailureRuns: number;
727
- /** Success fraction when every run or fallback trace row has a classified result. */
728
- reliabilityRate: number | null;
729
- passRate: number | null;
730
- meanScore: number | null;
731
- searchMeanScore: number | null;
732
- holdoutMeanScore: number | null;
733
- overfitGap: number | null;
734
- meanCostUsd: number | null;
735
- p95WallMs: number | null;
736
- failedRows: number;
737
- failuresWithAsi: number;
738
- singleShotTraces: number;
739
- multiShotTraces: number;
740
- splitCounts: Record<DatasetSplit, number>;
741
- domainCounts: Record<string, number>;
742
- failureClassCounts: Partial<Record<FailureClass, number>>;
743
- responsibleSurfaceCounts: Record<string, number>;
744
- /**
745
- * Runs excluded from `passRate` because the authenticity gate flagged them as
746
- * gamed. Surfaced, never silent: a release whose pass rate is computed over a
747
- * shrunken denominator has to say by how much, or the exclusion is just a
748
- * different way of hiding the same runs.
749
- */
750
- realnessGatedRuns: number;
751
- }
752
- interface ReleaseConfidenceScorecard {
753
- target: string;
754
- candidateId: string | null;
755
- baselineId: string | null;
756
- status: ReleaseConfidenceStatus;
757
- promote: boolean;
758
- axes: ReleaseConfidenceAxis[];
759
- issues: ReleaseConfidenceIssue[];
760
- metrics: ReleaseConfidenceMetrics;
761
- dataset: DatasetManifest | null;
762
- gateDecision: GateDecision | null;
763
- summary: string;
764
- }
765
- declare function evaluateReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
766
- declare function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
767
-
768
- interface RenderReleaseReportOptions {
769
- title?: string;
770
- runs?: readonly RunRecord[];
771
- comparator?: string;
772
- traceAnalystFindings?: readonly string[];
773
- nextActions?: readonly string[];
774
- }
775
- declare function renderReleaseReport(scorecard: ReleaseConfidenceScorecard, options?: RenderReleaseReportOptions): string;
776
-
777
- /**
778
- * Always-valid sequential evaluation.
779
- *
780
- * `researchReport` assumes a single pre-specified analysis. Real
781
- * consumers run campaigns weekly / nightly / per-PR; each new run silently
782
- * inflates the false-discovery rate, because the BH-FDR guarantee is for
783
- * the *first* look, not the 47th. Without time-uniform inference,
784
- * launch-decision teams either (a) don't peek, which forfeits the cost
785
- * advantage of stop-when-decisive, or (b) peek and pretend they didn't,
786
- * which forfeits scientific validity.
787
- *
788
- * This module ships **e-value-based confidence sequences** for paired
789
- * bounded outcomes. The methodology is the predictable plug-in betting
790
- * martingale of Waudby-Smith & Ramdas (2024) — provably valid at *any*
791
- * stopping time. Concretely:
792
- *
793
- * For paired deltas D_1, D_2, … ∈ [-c, c] with the null H_0: E[D] ≤ 0,
794
- * a betting fraction λ_i is chosen using only D_{1..i-1} (predictable
795
- * plug-in), and the running e-value is
796
- *
797
- * E_t = ∏_{i=1}^{t} (1 + λ_i · D_i)
798
- *
799
- * E_t is a non-negative martingale under H_0 with E[E_t] ≤ 1, so by
800
- * Ville's inequality, P(∃ t : E_t ≥ 1/α) ≤ α — we can reject the null
801
- * at any time without inflating the type-I error.
802
- *
803
- * Combined with `runEvalCampaign`, every consumer running rolling
804
- * campaigns gains the ability to ship the moment evidence is decisive,
805
- * stop-early on dead-on-arrival variants, and accumulate evidence across
806
- * partial runs without spending the FDR budget. No new sweep is wasted.
807
- *
808
- * References:
809
- * - Howard, S. R., Ramdas, A., McAuliffe, J., Sekhon, J. (2021).
810
- * Time-uniform, nonparametric, nonasymptotic confidence sequences.
811
- * Annals of Statistics, 49(2), 1055–1080.
812
- * - Waudby-Smith, I., Ramdas, A. (2024). Estimating means of bounded
813
- * random variables by betting. JRSS B, 86(1), 1–27.
814
- */
815
- type SequentialDecision = 'promote_now' | 'continue' | 'reject_now' | 'equivalent';
816
- interface PairedEvalueOptions {
817
- /**
818
- * Bound on |delta|. Default 1 (matching most score scales). Must satisfy
819
- * c > 0; deltas outside [-c, c] are clipped with a warning attached to
820
- * the return value.
821
- */
822
- bound?: number;
823
- /** Target Type-I error. Default 0.05. */
824
- alpha?: number;
825
- /**
826
- * Region of Practical Equivalence on the *mean* paired delta. When
827
- * supplied, the verdict can return `'equivalent'` once the running
828
- * confidence sequence on the mean is fully contained in [low, high].
829
- */
830
- rope?: {
831
- low: number;
832
- high: number;
833
- };
834
- /** Initial bet shrinkage (0 < scale ≤ 1). Default 0.5 — empirically robust. */
835
- initialBetShrinkage?: number;
836
- }
837
- interface PairedEvalueStep {
838
- /** 1-indexed observation count. */
839
- t: number;
840
- delta: number;
841
- /** Running e-value E_t = ∏ (1 + λ_i · D_i). */
842
- evalue: number;
843
- /** Time-uniform p-value at stopping time t. */
844
- pValue: number;
845
- /** Lower bound of the empirical Bernstein confidence sequence at level 1-α. */
846
- csLow: number;
847
- csHigh: number;
848
- /** Verdict at this stopping time. */
849
- decision: SequentialDecision;
850
- }
851
- interface PairedEvalueSequence {
852
- steps: PairedEvalueStep[];
853
- /** The decision at the final step. */
854
- finalDecision: SequentialDecision;
855
- /** Index (1-based) at which a non-`continue` decision first fired, or null. */
856
- decisionFiredAt: number | null;
857
- /** True if any deltas were clipped to [-bound, bound]. */
858
- clipped: boolean;
859
- }
860
- /**
861
- * Run the paired e-value sequence over an in-order delta stream.
862
- *
863
- * Use for *streaming* / interim analyses: pass the deltas you have so
864
- * far, get the verdict at every prefix length. The decision is
865
- * monotone-stable in the sense that once `'reject_now'` or `'promote_now'`
866
- * fires, the verdict at later steps remains decisive (the e-value is a
867
- * non-negative martingale; once it crosses the threshold, it's crossed).
868
- */
869
- declare function pairedEvalueSequence(deltas: number[], opts?: PairedEvalueOptions): PairedEvalueSequence;
870
- interface InterimReleaseConfidenceInput {
871
- /**
872
- * One delta series per candidate (paired deltas vs comparator). Order
873
- * within a series is the order the campaigns were run.
874
- */
875
- deltaSeries: Array<{
876
- candidateId: string;
877
- deltas: number[];
878
- }>;
879
- alpha?: number;
880
- bound?: number;
881
- rope?: {
882
- low: number;
883
- high: number;
884
- };
885
- }
886
- interface InterimReleaseConfidence {
887
- candidates: Array<{
888
- candidateId: string;
889
- decision: SequentialDecision;
890
- decisionFiredAt: number | null;
891
- finalEvalue: number;
892
- finalPValue: number;
893
- pairs: number;
894
- csLow: number;
895
- csHigh: number;
896
- }>;
897
- /**
898
- * Campaign-level recommendation: pick the strongest 'promote_now', else
899
- * 'continue' if any candidate is still live, else 'reject_now' if every
900
- * candidate is dead, else 'equivalent'.
901
- */
902
- recommendation: {
903
- decision: SequentialDecision;
904
- candidateId: string | null;
905
- };
906
- }
907
- /**
908
- * Run interim sequential analyses across many candidates at once,
909
- * preserving the time-uniform α guarantee for each candidate's series and
910
- * synthesising a campaign-level recommendation. Designed to be called on
911
- * every campaign tick — the recommendation is anytime-valid.
912
- */
913
- declare function evaluateInterimReleaseConfidence(input: InterimReleaseConfidenceInput): InterimReleaseConfidence;
914
-
915
- /**
916
- * Wilcoxon signed-rank test — paired non-parametric alternative.
917
- * Use when the differences aren't normally distributed.
918
- */
919
- declare function wilcoxonSignedRank(before: number[], after: number[]): {
920
- w: number;
921
- p: number;
922
- };
923
- /**
924
- * Benjamini–Hochberg false discovery rate. Returns adjusted q-values and
925
- * significance at the target FDR; handles ties and preserves q monotonicity.
926
- */
927
- declare function benjaminiHochberg(pValues: number[], fdr?: number): {
928
- qValues: number[];
929
- significant: boolean[];
930
- };
931
- interface PairedBootstrapResult {
932
- /** Number of paired observations. */
933
- n: number;
934
- /** Median of paired deltas (after − before). */
935
- median: number;
936
- /** Mean of paired deltas. */
937
- mean: number;
938
- /** Lower bound of the bootstrap CI on the chosen statistic. */
939
- low: number;
940
- /** Upper bound of the bootstrap CI on the chosen statistic. */
941
- high: number;
942
- /** Confidence level used (e.g. 0.95). */
943
- confidence: number;
944
- /** Number of bootstrap resamples used. */
945
- resamples: number;
946
- }
947
- interface PairedBootstrapOptions {
948
- /** Confidence level. Default 0.95. */
949
- confidence?: number;
950
- /** Bootstrap resample count. Default 2000. */
951
- resamples?: number;
952
- /** Statistic to bootstrap. Default 'median'. */
953
- statistic?: 'median' | 'mean';
954
- /** Deterministic seed. If omitted, uses Math.random(). */
955
- seed?: number;
956
- }
957
- /**
958
- * Paired bootstrap on (after − before) deltas. Returns a CI on the chosen
959
- * statistic (median by default); pairs are resampled with replacement. The
960
- * lower bound is what the promotion gate checks — `low > threshold` means the
961
- * gain is real at the confidence level. Throws on unequal sample sizes.
962
- */
963
- declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
964
-
965
- /**
966
- * FailureClusterView — groups failed runs by (failureClass, triggerTool,
967
- * argHash-prefix) so weekly reviews can prioritize the top-N clusters.
968
- *
969
- * Each cluster includes: N runs, scenarios affected, representative
970
- * error message, a proposed mitigation hint (rule → action table).
971
- */
972
-
973
- interface FailureCluster {
974
- failureClass: FailureClass;
975
- /** Tool name when the trigger was a tool span, else undefined. */
976
- toolName?: string;
977
- /** First 16 chars of argHash — clusters similar args. */
978
- argPrefix?: string;
979
- /**
980
- * Source dimension when the trigger was a judge span (e.g. `'format'`,
981
- * `'safety'`, `'correctness'`). Lets cross-template aggregators
982
- * group failures by the dimension that fired without overloading
983
- * `argPrefix`. Optional — clusters without this field deserialize cleanly.
984
- */
985
- dimension?: string;
986
- runCount: number;
987
- scenarioIds: string[];
988
- exampleError?: string;
989
- exampleRunId: string;
990
- }
991
- interface FailureClusterReport {
992
- clusters: FailureCluster[];
993
- totalFailures: number;
994
- totalRuns: number;
995
- }
996
-
997
- /**
998
- * Reporting helpers — production summaries and paper-quality figures — sit alongside `reporter.ts` rather
999
- * than replacing it.
1000
- *
1001
- * Three artefacts:
1002
- *
1003
- * - `summaryTable` Markdown table of per-candidate means,
1004
- * 95% bootstrap CIs, BH-adjusted Wilcoxon
1005
- * p-values, and Cohen's d versus a
1006
- * comparator candidate.
1007
- * - `paretoChart` Abstract spec for a cost vs quality
1008
- * scatter, with gate decisions overlaid.
1009
- * Returns numbers + labels — caller
1010
- * chooses the plotting library.
1011
- * - `gainHistogram`
1012
- * Per-item paired holdout deltas as a
1013
- * histogram spec (bins + counts + median +
1014
- * CI). Same "data, not images" contract.
1015
- *
1016
- * The figure types are PlotSpecs — JSON-friendly, library-agnostic.
1017
- * They aren't React components and they aren't PNGs; they are
1018
- * what you'd hand to vega-lite, plotly, matplotlib, or your own
1019
- * Canvas renderer to draw the actual figure.
1020
- */
1021
-
1022
- interface SummaryTableOptions {
1023
- /** Comparator candidate id. Wilcoxon + paired Cohen's dz are computed
1024
- * versus this candidate. Required for paired stats columns. */
1025
- comparator?: string;
1026
- /** Which split to read scores from. Default 'holdout'. */
1027
- split?: 'search' | 'holdout';
1028
- /** Confidence level for the bootstrap CI on the mean. Default 0.95. */
1029
- confidence?: number;
1030
- /** FDR for BH adjustment of the comparison p-values. Default 0.05. */
1031
- fdr?: number;
1032
- }
1033
- interface SummaryTableRow {
1034
- candidateId: string;
1035
- n: number;
1036
- mean: number;
1037
- ciLow: number;
1038
- ciHigh: number;
1039
- /** BH-adjusted q-value vs comparator, or null when unavailable. */
1040
- qValue: number | null;
1041
- /** Paired Cohen's dz vs comparator, or null when the paired variance is zero. */
1042
- cohensD: number | null;
1043
- /** Matched observations used for paired comparison, or null on the comparator row. */
1044
- pairedN: number | null;
1045
- /** Candidate observations without a comparator match. */
1046
- unpairedCandidateN: number | null;
1047
- /** Comparator observations without a candidate match. */
1048
- unpairedComparatorN: number | null;
1049
- }
1050
- interface SummaryTable {
1051
- rows: SummaryTableRow[];
1052
- comparator: string | null;
1053
- split: 'search' | 'holdout';
1054
- /** Pre-rendered markdown — drop into a paper or PR. */
1055
- markdown: string;
1056
- }
1057
- /**
1058
- * Table 1 helper. Buckets runs by `candidateId`, computes mean +
1059
- * bootstrap CI on the chosen split, and (when a comparator is given)
1060
- * BH-adjusted Wilcoxon p + paired Cohen's dz versus that comparator.
1061
- */
1062
- declare function summaryTable(runs: RunRecord[], opts?: SummaryTableOptions): SummaryTable;
1063
- interface ParetoPoint {
1064
- candidateId: string;
1065
- /** Mean USD cost per run on the chosen split. */
1066
- cost: number;
1067
- /** Mean score on the chosen split. */
1068
- quality: number;
1069
- /** Number of runs that informed this point. */
1070
- n: number;
1071
- /** Whether this candidate is on the Pareto frontier — high
1072
- * quality, low cost, no dominator. */
1073
- onFrontier: boolean;
1074
- /** Optional gate verdict for this candidate, if a `GateDecision`
1075
- * for it was passed in. */
1076
- gate?: 'promote' | 'reject';
1077
- }
1078
- interface ParetoFigureSpec {
1079
- kind: 'pareto-cost-quality';
1080
- split: 'search' | 'holdout';
1081
- points: ParetoPoint[];
1082
- axes: {
1083
- x: 'costUsd';
1084
- y: 'score';
1085
- };
1086
- }
1087
- /**
1088
- * Cost vs quality scatter spec. `gateDecisions` is keyed by
1089
- * candidate id; if present, every point picks up the gate verdict
1090
- * for overlay.
1091
- */
1092
- declare function paretoChart(runs: RunRecord[], opts?: {
1093
- split?: 'search' | 'holdout';
1094
- gateDecisions?: Record<string, GateDecision>;
1095
- }): ParetoFigureSpec;
1096
- interface GainDistributionBin {
1097
- /** Inclusive lower edge. */
1098
- lo: number;
1099
- /** Exclusive upper edge (or inclusive if it's the last bin). */
1100
- hi: number;
1101
- /** Number of pairs whose delta lands in this bin. */
1102
- count: number;
1103
- }
1104
- interface GainDistributionFigureSpec {
1105
- kind: 'gain-distribution';
1106
- candidateId: string;
1107
- comparator: string;
1108
- split: 'search' | 'holdout';
1109
- /** Number of pairs used. */
1110
- n: number;
1111
- /** Candidate rows without a comparator match. */
1112
- unpairedCandidateN: number;
1113
- /** Comparator rows without a candidate match. */
1114
- unpairedComparatorN: number;
1115
- bins: GainDistributionBin[];
1116
- median: number | null;
1117
- ci: {
1118
- low: number;
1119
- high: number;
1120
- } | null;
1121
- }
1122
- interface GainDistributionOptions {
1123
- /** Number of histogram bins. Default 11 (so the centre is exact at 0). */
1124
- bins?: number;
1125
- /** Which split to use. Default 'holdout'. */
1126
- split?: 'search' | 'holdout';
1127
- /** Confidence level for the CI. Default 0.95. */
1128
- confidence?: number;
1129
- /** Bootstrap resamples. Default 2000. */
1130
- resamples?: number;
1131
- /** Deterministic seed. */
1132
- seed?: number;
1133
- }
1134
- /**
1135
- * Held-out improvement distribution: per-pair delta (candidate −
1136
- * comparator), histogrammed. Includes the bootstrap CI on the median
1137
- * delta — same primitive the promotion gate uses.
1138
- */
1139
- declare function gainHistogram(runs: RunRecord[], candidateId: string, comparator: string, opts?: GainDistributionOptions): GainDistributionFigureSpec;
1140
- type ResearchReportDecision = 'promote' | 'hold' | 'reject' | 'equivalent' | 'needs_more_data';
1141
- /**
1142
- * Hard floor below which a paired comparison is treated as uninformative
1143
- * regardless of `minPairs`. Mirrors the lower limit on Wilcoxon signed-rank
1144
- * exact tables; below this the test has no power to separate effect sizes.
1145
- */
1146
- declare const RESEARCH_REPORT_HARD_PAIR_FLOOR = 6;
1147
- interface ResearchReportOptions {
1148
- /** Human-readable report title. */
1149
- title?: string;
1150
- /** Comparator candidate id. Required for statistical decision guidance. */
1151
- comparator?: string;
1152
- /** Which split to use for the primary decision. Default 'holdout'. */
1153
- split?: 'search' | 'holdout';
1154
- /** Confidence level used by lower-level report helpers. Default 0.95. */
1155
- confidence?: number;
1156
- /** FDR threshold for q-values. Default 0.05. */
1157
- fdr?: number;
1158
- /**
1159
- * Soft floor on paired observations before issuing a directional
1160
- * promote / reject. Below this we report `needs_more_data` and surface the
1161
- * minimum detectable effect at the current N. Default 20 — chosen so the
1162
- * Wilcoxon signed-rank approximation is reasonable and so the paired
1163
- * bootstrap CI has non-degenerate coverage. Hard floor is enforced at
1164
- * `RESEARCH_REPORT_HARD_PAIR_FLOOR` (6) regardless of this value.
1165
- */
1166
- minPairs?: number;
1167
- /**
1168
- * Region of Practical Equivalence on the paired delta. When a candidate's
1169
- * paired-delta CI is fully contained in `[low, high]`, the decision is
1170
- * `equivalent` rather than `hold`. Sourced from the domain owner — there is
1171
- * no statistically-defensible default.
1172
- */
1173
- rope?: {
1174
- low: number;
1175
- high: number;
1176
- };
1177
- /**
1178
- * Power for the minimum detectable effect (MDE) reported on each candidate.
1179
- * Default 0.8.
1180
- */
1181
- mdePower?: number;
1182
- /**
1183
- * Two-sided alpha for the MDE. Default matches `fdr` so the reported MDE
1184
- * lines up with the test the report actually runs.
1185
- */
1186
- mdeAlpha?: number;
1187
- /** Optional held-out gate decisions keyed by candidate id. */
1188
- gateDecisions?: Record<string, GateDecision>;
1189
- /** Optional failure clusters from failureClusterView. */
1190
- failureClusters?: FailureClusterReport;
1191
- /** Build gain histograms for these candidates. Defaults to all non-comparator candidates. */
1192
- candidateIds?: string[];
1193
- /** Deterministic bootstrap seed passed to gainHistogram and the posterior helper. */
1194
- seed?: number;
1195
- /** Report timestamp. Defaults to current time. */
1196
- generatedAt?: string;
1197
- /**
1198
- * Hash of a preregistered protocol (e.g. `signManifest({...}).contentHash`).
1199
- * Embedded verbatim in the report so the analysis can be cited as the
1200
- * preregistered one rather than a post-hoc fishing expedition.
1201
- */
1202
- preregistrationHash?: string;
1203
- }
1204
- interface ResearchReportRecommendation {
1205
- decision: ResearchReportDecision;
1206
- candidateId: string | null;
1207
- rationale: string[];
1208
- risks: string[];
1209
- nextActions: string[];
1210
- }
1211
- interface ResearchReportCandidate {
1212
- candidateId: string;
1213
- n: number;
1214
- mean: number;
1215
- ciLow: number;
1216
- ciHigh: number;
1217
- qValue: number | null;
1218
- cohensD: number | null;
1219
- meanDeltaVsComparator: number | null;
1220
- pairedN: number;
1221
- medianGain: number | null;
1222
- meanGain: number | null;
1223
- gainCi: {
1224
- low: number;
1225
- high: number;
1226
- } | null;
1227
- /**
1228
- * Bayesian-bootstrap posterior summaries on the paired mean delta.
1229
- * Dirichlet(1, ..., 1) weights represent uncertainty over the empirical
1230
- * distribution of matched deltas.
1231
- */
1232
- prGreaterThanZero: number | null;
1233
- prInRope: number | null;
1234
- /**
1235
- * Minimum detectable effect (in score units) at the candidate's paired N,
1236
- * the configured power, and the configured alpha. Standardised by the
1237
- * observed paired-delta SD and inverted via `requiredSampleSize`. Reported
1238
- * for every candidate so a `needs_more_data` verdict is actionable.
1239
- */
1240
- mde: number | null;
1241
- onParetoFrontier: boolean;
1242
- gate?: ParetoPoint['gate'];
1243
- decision: ResearchReportDecision;
1244
- decisionReason: string;
1245
- }
1246
- interface ResearchReportMethodology {
1247
- /**
1248
- * Plain-language assumptions the report depends on. Read these first when
1249
- * deciding whether the verdict is load-bearing for a launch decision.
1250
- */
1251
- assumptions: string[];
1252
- /** Tests and estimators the verdict was computed from. */
1253
- methods: string[];
1254
- /** Alternatives the author considered and why this report didn't take them. */
1255
- alternatives: string[];
1256
- /** Failure modes — when this report should NOT drive a decision. */
1257
- whenNotToApply: string[];
1258
- /** Citations for the methodological choices above. */
1259
- citations: string[];
1260
- }
1261
- interface ResearchReport {
1262
- kind: 'agent-eval-research-report';
1263
- title: string;
1264
- generatedAt: string;
1265
- split: 'search' | 'holdout';
1266
- comparator: string | null;
1267
- /**
1268
- * SHA-256 over the canonicalised set of `(runId, candidateId, split)` triples
1269
- * the report was computed from, plus the comparator and split. Stable across
1270
- * key insertion order; recomputable by the reader to verify provenance.
1271
- */
1272
- runFingerprint: string;
1273
- preregistrationHash: string | null;
1274
- rope: {
1275
- low: number;
1276
- high: number;
1277
- } | null;
1278
- executiveSummary: string[];
1279
- recommendation: ResearchReportRecommendation;
1280
- candidates: ResearchReportCandidate[];
1281
- summary: SummaryTable;
1282
- charts: {
1283
- pareto: ParetoFigureSpec;
1284
- gains: GainDistributionFigureSpec[];
1285
- };
1286
- methodology: ResearchReportMethodology;
1287
- failureClusters?: FailureClusterReport;
1288
- markdown: string;
1289
- html: string;
1290
- }
1291
- /**
1292
- * Executive research report for CPO / AI-lead / launch-review consumption.
1293
- *
1294
- * Composes:
1295
- * - `summaryTable` marginal stats with BH-FDR-adjusted q-values
1296
- * - `paretoChart` cost-vs-quality frontier with gate overlay
1297
- * - `gainHistogram` per-candidate paired-delta distribution
1298
- * - paired posterior (this file): bootstrap CI on median, Bayesian-bootstrap Pr(Δ>0),
1299
- * Pr(Δ∈ROPE), MDE at the configured power
1300
- *
1301
- * Decisions are made on paired evidence — never on marginal means alone —
1302
- * and respect any held-out gate decision the caller passes through. The
1303
- * report embeds a SHA-256 fingerprint of the input run set and, optionally,
1304
- * the hash of a preregistered protocol so a downstream reader can verify
1305
- * provenance and that the analysis was the preregistered one.
1306
- *
1307
- * Async because the fingerprint uses Web Crypto via `hashJson`; deterministic
1308
- * for any fixed `runs`, `seed`, and ROPE.
1309
- */
1310
- declare function researchReport(runs: RunRecord[], opts?: ResearchReportOptions): Promise<ResearchReport>;
1311
-
1312
- export { type BootstrapOptions, type BootstrapResult, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeReplayGateArgs, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type ParetoFigureSpec, type ParetoPoint, RESEARCH_REPORT_HARD_PAIR_FLOOR, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type RubricOutcomePair, type RubricPredictiveValidityInput, type RubricPredictiveValidityReport, type RubricRanking, type SequentialDecision, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type Verdict, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, renderReleaseReport, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
1
+ import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "./rubric-predictive-validity-Ku_clp_1.js";
2
+ import { I as pairedBootstrap, Z as wilcoxonSignedRank, d as PairedBootstrapOptions, f as PairedBootstrapResult, y as benjaminiHochberg } from "./statistics-Cmj6nynr.js";
3
+ import { _ as paretoChart, a as ParetoPoint, c as ResearchReportCandidate, d as ResearchReportOptions, f as ResearchReportRecommendation, g as gainHistogram, h as SummaryTableRow, i as ParetoFigureSpec, l as ResearchReportDecision, m as SummaryTableOptions, n as GainDistributionFigureSpec, o as RESEARCH_REPORT_HARD_PAIR_FLOOR, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, u as ResearchReportMethodology, v as researchReport, y as summaryTable } from "./summary-report-Cj9gdw4i.js";
4
+ import { _ as ReleaseConfidenceStatus, a as JudgeReplayGateArgs, b as assertReleaseConfidence, c as judgeReplayGate, d as ReleaseConfidenceAxis, f as ReleaseConfidenceAxisName, g as ReleaseConfidenceScorecard, h as ReleaseConfidenceMetrics, i as BootstrapResult, m as ReleaseConfidenceIssue, n as renderReleaseReport, o as Verdict, p as ReleaseConfidenceInput, r as BootstrapOptions, s as bootstrapCi, t as RenderReleaseReportOptions, v as ReleaseConfidenceThresholds, x as evaluateReleaseConfidence, y as ReleaseTraceEvidence } from "./release-report-mvB2G4_J.js";
5
+ import { a as PairedEvalueStep, c as pairedEvalueSequence, i as PairedEvalueSequence, n as InterimReleaseConfidenceInput, o as SequentialDecision, r as PairedEvalueOptions, s as evaluateInterimReleaseConfidence, t as InterimReleaseConfidence } from "./sequential-CYwq6Ff_.js";
6
+ export { type BootstrapOptions, type BootstrapResult, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeReplayGateArgs, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type ParetoFigureSpec, type ParetoPoint, RESEARCH_REPORT_HARD_PAIR_FLOOR, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type RubricOutcomePair, type RubricPredictiveValidityInput, type RubricPredictiveValidityReport, type RubricRanking, type SequentialDecision, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type Verdict, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, renderReleaseReport, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };