@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
package/dist/fuzz.d.ts CHANGED
@@ -1,285 +1,21 @@
1
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
2
- interface CostUsage {
3
- inputTokens: number;
4
- /** Includes reasoning tokens when the provider bills them as output. */
5
- outputTokens: number;
6
- /** Reasoning-token subset of outputTokens, when reported. */
7
- reasoningTokens?: number;
8
- /** Prompt tokens served from a provider cache. */
9
- cachedTokens?: number;
10
- /** Prompt tokens written into a provider cache. */
11
- cacheWriteTokens?: number;
12
- }
13
- interface CostCallBase {
14
- callId: string;
15
- channel: CostChannel;
16
- phase: string;
17
- actor: string;
18
- model: string;
19
- maximumCostUsd?: number;
20
- tags?: Record<string, string>;
21
- timestamp: number;
22
- }
23
- interface PendingCostCall extends CostCallBase {
24
- status: 'pending';
25
- }
26
- interface PendingCostCallView extends PendingCostCall {
27
- state: 'active' | 'late' | 'interrupted';
28
- }
29
- interface CostReceipt extends CostCallBase, CostUsage {
30
- status: 'settled';
31
- costUsd: number;
32
- costUnknown: boolean;
33
- usageUnknown?: boolean;
34
- /** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
35
- pricing?: {
36
- inputUsdPerThousand: number;
37
- cachedInputUsdPerThousand?: number;
38
- cacheWriteUsdPerThousand?: number;
39
- outputUsdPerThousand: number;
40
- };
41
- /** Cost reported by the provider, not a local token-price calculation. */
42
- actualCostUsd?: number;
43
- error?: string;
44
- }
45
- interface CostReceiptInput extends CostUsage {
46
- model: string;
47
- /** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
48
- customTokenPricing?: CustomTokenPricing;
49
- actualCostUsd?: number;
50
- costUnknown?: boolean;
51
- usageUnknown?: boolean;
52
- }
53
- /** Per-million token rates for a model or endpoint not covered by package pricing. */
54
- interface CustomTokenPricing {
55
- /** Non-cached input tokens. */
56
- inputUsdPerMillion: number;
57
- /** Cache-read tokens. Falls back to the normal input rate when omitted. */
58
- cachedInputUsdPerMillion?: number;
59
- /** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
60
- cacheWriteUsdPerMillion?: number;
61
- outputUsdPerMillion: number;
62
- }
63
- type MaximumCharge = {
64
- externallyEnforcedMaximumUsd: number;
65
- } | ({
66
- customTokenPricing: CustomTokenPricing;
67
- } & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
68
- model: string;
69
- } & CostUsage);
70
- interface RunPaidCallInput<T> {
71
- callId?: string;
72
- channel: CostChannel;
73
- phase: string;
74
- actor: string;
75
- /** Used before a provider receipt exists and on failures without one. */
76
- model?: string;
77
- tags?: Record<string, string>;
78
- signal?: AbortSignal;
79
- /** Provider-enforced dollar maximum, or maximum token usage with known pricing. Required when capped. */
80
- maximumCharge?: MaximumCharge;
81
- /** `callId` can be forwarded as the provider's idempotency key. */
82
- execute(signal: AbortSignal, callId: string): Promise<T>;
83
- receipt(value: T): CostReceiptInput;
84
- receiptFromError?(error: Error): CostReceiptInput | undefined;
85
- }
86
- type PaidCallResult<T> = {
87
- succeeded: true;
88
- callId: string;
89
- value: T;
90
- receipt: CostReceipt;
91
- } | {
92
- succeeded: false;
93
- callId?: string;
94
- error: Error;
95
- receipt?: CostReceipt;
96
- };
97
- interface ChannelRollup {
98
- channel: CostChannel;
99
- calls: number;
100
- inputTokens: number;
101
- outputTokens: number;
102
- reasoningTokens?: number;
103
- cachedTokens: number;
104
- cacheWriteTokens?: number;
105
- costUsd: number;
106
- unpricedCalls: number;
107
- unknownUsageCalls: number;
108
- }
109
- interface CostLedgerSummary {
110
- totalCalls: number;
111
- pendingCalls: number;
112
- unresolvedCalls: number;
113
- reservedCostUsd: number;
114
- inputTokens: number;
115
- outputTokens: number;
116
- reasoningTokens?: number;
117
- cachedTokens: number;
118
- cacheWriteTokens?: number;
119
- totalCostUsd: number;
120
- byChannel: ChannelRollup[];
121
- unpricedModels: string[];
122
- fullyPriced: boolean;
123
- usageComplete: boolean;
124
- accountingComplete: boolean;
125
- incompleteReasons: string[];
126
- }
127
- interface CostLedgerFilter {
128
- channel?: CostChannel;
129
- phase?: string;
130
- tags?: Record<string, string>;
131
- }
132
- interface CostLedgerWaitOptions {
133
- /** Maximum time to wait for active provider calls. Default 5 seconds. */
134
- timeoutMs?: number;
135
- /** Wait only for calls matching this attribution filter. */
136
- filter?: CostLedgerFilter;
137
- }
138
- /** Append-only storage. `append` must atomically reject stale revisions. */
139
- interface CostLedgerPersistence {
140
- read(): {
141
- revision: string;
142
- events: string;
143
- };
144
- append(expectedRevision: string, event: string): string | undefined;
145
- }
146
- interface CostLedgerOptions {
147
- costCeilingUsd?: number;
148
- persistence?: CostLedgerPersistence;
149
- /** Import already-settled receipts without admitting new paid work. */
150
- receipts?: readonly CostReceipt[];
151
- }
152
- /** Run-wide paid-call admission, durable call state, receipts, and summaries. */
153
- declare class CostLedger {
154
- private readonly records;
155
- private readonly activeCallIds;
156
- private readonly lateCallIds;
157
- private readonly idleWaiters;
158
- private completedTasks;
159
- private revision;
160
- private costLimitPersisted;
161
- readonly costCeilingUsd?: number;
162
- private readonly persistence?;
163
- constructor(input?: number | CostLedgerOptions);
164
- runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
165
- /** Wait until every call started by this ledger has produced a durable outcome. */
166
- waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
167
- /** Settle a call left pending by a crashed process after reconciling with the provider. */
168
- reconcile(callId: string, observed: CostReceiptInput, options?: {
169
- error?: string;
170
- }): CostReceipt;
171
- list(filter?: CostLedgerFilter): CostReceipt[];
172
- /** Read pending calls without exposing mutable ledger state. */
173
- listPending(filter?: CostLedgerFilter): PendingCostCallView[];
174
- summary(filter?: CostLedgerFilter): CostLedgerSummary;
175
- markCompleted(count?: number): void;
176
- costPerCompletedTask(): number | null;
177
- private execute;
178
- private captureLateOutcome;
179
- private releaseActiveCall;
180
- private commitOutcome;
181
- private captureFailure;
182
- private commitReceipt;
183
- private resolveMaximum;
184
- private hasIncompleteSettledCall;
185
- private appendRecord;
186
- private ensureCostLimitPersisted;
187
- private appendEvent;
188
- }
189
- /** Public callback surface for a shared cost ledger.
190
- *
191
- * Declaration bundles may expose this type through multiple package subpaths.
192
- * Keeping callback contracts structural lets those subpaths compose while the
193
- * concrete {@link CostLedger} retains its private durable state.
194
- */
195
- type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending' | 'waitForIdle'>> & Partial<Pick<CostLedger, 'listPending' | 'waitForIdle'>>;
196
-
197
- /**
198
- * Adversarial mutation contract.
199
- *
200
- * `AdversarialMutation<S>` is the scenario-mutation strategy the fuzz harness
201
- * (`fuzzAgent`, src/fuzz) drives: paraphrase, edge-case substitution, or
202
- * compositional combination of a scenario the policy currently passes, looking
203
- * for the tail inputs that break it. The harness supplies the loop; consumers
204
- * supply the mutations and the failure detector.
205
- */
206
- interface AdversarialMutation<S> {
207
- id: string;
208
- /**
209
- * Mutate one scenario. Return null to skip; return one or more new
210
- * scenarios. The harness deduplicates by `mutateScenarioId(scenario)`.
211
- */
212
- mutate(parent: S, rng: () => number): Promise<S[]> | S[];
213
- }
214
-
215
- /**
216
- * Validator-output verdict — substrate primitive for "did this output pass,
217
- * and how well?"
218
- *
219
- * Used by:
220
- * - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
221
- * - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
222
- * Runtime keeps `Validator` because it's coupled to runtime-shaped
223
- * `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
224
- * itself is a substrate concept and lives here.
225
- *
226
- * Repo layering: agent-eval is the substrate (no upward deps). Both
227
- * agent-runtime and agent-knowledge consume this type FROM agent-eval —
228
- * never the other way around. See CLAUDE.md "Repo layering" for the rule.
229
- */
230
- /**
231
- * Minimal verdict shape — `valid` + `score` are required; `scores` +
232
- * `notes` are optional surface. Validators that need richer shapes
233
- * parameterise `Validator<Output, MyVerdict>` with their own type.
234
- *
235
- * Need structured extras? Extend DefaultVerdict with typed fields — never
236
- * serialize extras into `notes`.
237
- */
238
- interface DefaultVerdict {
239
- /** Whether the output meets the validator's pass criteria. */
240
- valid: boolean;
241
- /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
242
- score: number;
243
- /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
244
- scores?: Record<string, number>;
245
- /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
246
- notes?: string;
247
- }
248
-
249
- /**
250
- * Behavior-space exploration — types.
251
- *
252
- * One engine searches a space of inputs against a target, scores each run with a
253
- * multi-objective verdict, keeps a quality-diversity archive, and admits only
254
- * findings that pass the validity gates. Adversarial fuzzing is the headline
255
- * preset (`fuzzAgent`); swapping the `Objective` re-points the same engine at
256
- * novelty search, curriculum growth, or user-simulation.
257
- *
258
- * Two kinds of coordinates, deliberately distinct:
259
- * - INPUT axes (`space.axes`) are the stratification plan — enumerable up front,
260
- * so allocation and the coverage denominator (planned vs covered) are honest.
261
- * - MEASURED descriptors (`descriptor(scenario, ev)`) are read off the rollout —
262
- * they bin the archive by what the agent DID, and never inflate the coverage
263
- * denominator (a behavior you haven't seen yet is not a planned cell).
264
- *
265
- * An `Evaluation` IS a `DefaultVerdict` — same spine as judges and verifiers,
266
- * never a parallel score shape.
267
- */
268
-
1
+ import { t as DefaultVerdict } from "./verdict-Dps8_okt.js";
2
+ import { a as CostChannel, b as MaximumCharge, c as CostLedgerHandle } from "./cost-ledger-Dye6jCgg.js";
3
+ import { t as AdversarialMutation } from "./adversarial-smnADNFS.js";
4
+ //#region src/fuzz/types.d.ts
269
5
  /** One input axis of the stratification plan (e.g. matterType, difficulty, personaRigor). */
270
6
  interface SpaceAxis {
271
- name: string;
272
- values: string[];
7
+ name: string;
8
+ values: string[];
273
9
  }
274
10
  /** The input space to stratify. Cells are the cartesian product of the axes. */
275
11
  interface BehaviorSpace {
276
- axes: SpaceAxis[];
12
+ axes: SpaceAxis[];
277
13
  }
278
14
  /** One input cell: a coordinate in the stratification plan. */
279
15
  interface Cell {
280
- /** Stable id, e.g. `matterType=nda|difficulty=hard`. */
281
- id: string;
282
- coords: Record<string, string>;
16
+ /** Stable id, e.g. `matterType=nda|difficulty=hard`. */
17
+ id: string;
18
+ coords: Record<string, string>;
283
19
  }
284
20
  /**
285
21
  * The outcome of running the target against one scenario: a `DefaultVerdict`
@@ -288,31 +24,31 @@ interface Cell {
288
24
  * WHICH dimension is weak only when evaluations carry it.
289
25
  */
290
26
  interface Evaluation extends DefaultVerdict {
291
- /** Measured behavior coordinates (e.g. `{ outcome: 'refused' }`). Bins the archive. */
292
- descriptor?: Record<string, string>;
293
- /** Surfaced output — drives exemplars + minimization. */
294
- output?: string;
295
- /** RunRecord id when the target persisted a trace. */
296
- runId?: string;
297
- /** Structured labels, e.g. failure classes (`hallucination`, `refusal`). */
298
- labels?: string[];
299
- /** Wall-clock for the evaluation, when the consumer measures it more precisely
300
- * than the engine can (e.g. excluding judge time). Engine-measured otherwise. */
301
- latencyMs?: number;
27
+ /** Measured behavior coordinates (e.g. `{ outcome: 'refused' }`). Bins the archive. */
28
+ descriptor?: Record<string, string>;
29
+ /** Surfaced output — drives exemplars + minimization. */
30
+ output?: string;
31
+ /** RunRecord id when the target persisted a trace. */
32
+ runId?: string;
33
+ /** Structured labels, e.g. failure classes (`hallucination`, `refusal`). */
34
+ labels?: string[];
35
+ /** Wall-clock for the evaluation, when the consumer measures it more precisely
36
+ * than the engine can (e.g. excluding judge time). Engine-measured otherwise. */
37
+ latencyMs?: number;
302
38
  }
303
39
  /** Run the target against one scenario in a cell. */
304
40
  type Evaluator<S> = (scenario: S, cell: Cell) => Promise<Evaluation>;
305
41
  /** Context a proposer sees — prior elites + findings let a skill-backed proposer steer. */
306
42
  interface ProposeContext<S> {
307
- cell: Cell;
308
- seeds: S[];
309
- /** Current archive elites whose input cell matches. */
310
- elites: S[];
311
- /** Verified findings so far (read-only) — probe new gaps, not re-found ones. */
312
- findings: ReadonlyArray<Finding<S>>;
313
- /** How many candidates to propose. */
314
- count: number;
315
- rng: () => number;
43
+ cell: Cell;
44
+ seeds: S[];
45
+ /** Current archive elites whose input cell matches. */
46
+ elites: S[];
47
+ /** Verified findings so far (read-only) — probe new gaps, not re-found ones. */
48
+ findings: ReadonlyArray<Finding<S>>;
49
+ /** How many candidates to propose. */
50
+ count: number;
51
+ rng: () => number;
316
52
  }
317
53
  /**
318
54
  * Produces candidate scenarios for a cell. A plain function — `mutationProposer`
@@ -331,14 +67,14 @@ type MutationProposer<S> = (ctx: ProposeContext<S>) => Promise<S[]> | S[];
331
67
  * score is interesting) and `noveltyObjective` (far from the archive) ship.
332
68
  */
333
69
  interface Objective {
334
- kind: string;
335
- interest(ev: Evaluation, ctx: ObjectiveContext): number;
336
- /** Default 0.5. */
337
- threshold?: number;
70
+ kind: string;
71
+ interest(ev: Evaluation, ctx: ObjectiveContext): number;
72
+ /** Default 0.5. */
73
+ threshold?: number;
338
74
  }
339
75
  interface ObjectiveContext {
340
- archiveScores: number[];
341
- archiveDescriptors: Array<Record<string, string> | undefined>;
76
+ archiveScores: number[];
77
+ archiveDescriptors: Array<Record<string, string> | undefined>;
342
78
  }
343
79
  /**
344
80
  * Validity gates — the moat. A notable candidate is admitted ONLY when it is a
@@ -347,113 +83,113 @@ interface ObjectiveContext {
347
83
  * LLM); live wiring supplies real gates.
348
84
  */
349
85
  interface ValidityGates<S> {
350
- isValid?: (scenario: S, ev: Evaluation, cell: Cell) => boolean | Promise<boolean>;
351
- isUncontaminated?: (scenario: S, ev: Evaluation, cell: Cell) => boolean | Promise<boolean>;
86
+ isValid?: (scenario: S, ev: Evaluation, cell: Cell) => boolean | Promise<boolean>;
87
+ isUncontaminated?: (scenario: S, ev: Evaluation, cell: Cell) => boolean | Promise<boolean>;
352
88
  }
353
89
  /** A gate-verified, minimized finding — the unit the capsule reports. */
354
90
  interface Finding<S> {
355
- id: string;
356
- cell: Cell;
357
- scenario: S;
358
- /** The minimized trigger (== `scenario` when no minimizer is supplied). */
359
- minimized: S;
360
- /** Legible text of the minimized trigger, when `scenarioText` is supplied. */
361
- text?: string;
362
- /** The full multi-objective verdict. */
363
- evaluation: Evaluation;
364
- /** The objective's interest score that flagged it. */
365
- interest: number;
366
- /** Which objective flagged it. */
367
- objective: string;
91
+ id: string;
92
+ cell: Cell;
93
+ scenario: S;
94
+ /** The minimized trigger (== `scenario` when no minimizer is supplied). */
95
+ minimized: S;
96
+ /** Legible text of the minimized trigger, when `scenarioText` is supplied. */
97
+ text?: string;
98
+ /** The full multi-objective verdict. */
99
+ evaluation: Evaluation;
100
+ /** The objective's interest score that flagged it. */
101
+ interest: number;
102
+ /** Which objective flagged it. */
103
+ objective: string;
368
104
  }
369
105
  /** An archive elite — the most interesting scenario seen for one bin. */
370
106
  interface ArchiveEntry<S> {
371
- /** Input cell + measured descriptor coords combined, e.g. `difficulty=hard|outcome=refused`. */
372
- binId: string;
373
- cell: Cell;
374
- scenario: S;
375
- evaluation: Evaluation;
376
- interest: number;
107
+ /** Input cell + measured descriptor coords combined, e.g. `difficulty=hard|outcome=refused`. */
108
+ binId: string;
109
+ cell: Cell;
110
+ scenario: S;
111
+ evaluation: Evaluation;
112
+ interest: number;
377
113
  }
378
114
  /** Summary of a sample — every aggregate carries its spread, never a bare mean. */
379
115
  interface Distribution {
380
- mean: number;
381
- median: number;
382
- p90: number;
383
- min: number;
384
- max: number;
385
- n: number;
116
+ mean: number;
117
+ median: number;
118
+ p90: number;
119
+ min: number;
120
+ max: number;
121
+ n: number;
386
122
  }
387
123
  /** Per-INPUT-cell coverage — the planned-vs-covered map. */
388
124
  interface CoverageCell {
389
- cell: Cell;
390
- runs: number;
391
- /** Headline score distribution in [0,1]; `null` when the cell was never run
392
- * (honestly uncovered — never a fabricated zero). */
393
- score: Distribution | null;
394
- /** Fraction of runs the objective flagged as notable. */
395
- findingRate: number;
396
- /** Per-dimension score distributions — surfaces WHICH dimension is weak and
397
- * how consistently. */
398
- dimensions: Record<string, Distribution>;
399
- /** Evaluation wall-clock per run; engine-measured unless the evaluation
400
- * carried its own `latencyMs`. `null` when the cell was never run. */
401
- latencyMs: Distribution | null;
402
- /** Known dollars spent in this cell — present only when cost tracking was
403
- * wired; runs with unknown cost are counted apart, never folded in as $0. */
404
- costUsd?: number;
405
- costUnknownRuns?: number;
125
+ cell: Cell;
126
+ runs: number;
127
+ /** Headline score distribution in [0,1]; `null` when the cell was never run
128
+ * (honestly uncovered — never a fabricated zero). */
129
+ score: Distribution | null;
130
+ /** Fraction of runs the objective flagged as notable. */
131
+ findingRate: number;
132
+ /** Per-dimension score distributions — surfaces WHICH dimension is weak and
133
+ * how consistently. */
134
+ dimensions: Record<string, Distribution>;
135
+ /** Evaluation wall-clock per run; engine-measured unless the evaluation
136
+ * carried its own `latencyMs`. `null` when the cell was never run. */
137
+ latencyMs: Distribution | null;
138
+ /** Known dollars spent in this cell — present only when cost tracking was
139
+ * wired; runs with unknown cost are counted apart, never folded in as $0. */
140
+ costUsd?: number;
141
+ costUnknownRuns?: number;
406
142
  }
407
143
  /** The artifact every exploration produces. */
408
144
  interface CapsuleData<S> {
409
- target: string;
410
- objective: string;
411
- /** Stamped by the caller — the engine stays clock-free and deterministic. */
412
- generatedAt?: string;
413
- coverage: CoverageCell[];
414
- /** Verified findings, sorted by descending interest. */
415
- findings: Finding<S>[];
416
- /** QD archive elites (binned by input × measured coords). */
417
- archive: ArchiveEntry<S>[];
418
- /** Post-harden lift, filled by a second pass after an improvement. */
419
- lift?: {
420
- before: number;
421
- after: number;
422
- verdict: string;
423
- };
424
- stats: {
425
- totalRuns: number;
426
- /** Input-cell denominator — the stratification plan. */
427
- cellsTotal: number;
428
- cellsCovered: number;
429
- /** Distinct measured-descriptor bins observed (never part of the denominator). */
430
- behaviorBinsObserved: number;
431
- candidateFindings: number;
432
- verifiedFindings: number;
433
- /** Distribution of per-cell mean scores across covered cells (cells weigh
434
- * equally — variance steering sends more runs to weak cells, so a
435
- * run-weighted average would bias low). `null` when nothing ran. */
436
- robustness: Distribution | null;
437
- /** Evaluation wall-clock across all runs. `null` when nothing ran. */
438
- latencyMs: Distribution | null;
439
- /** Known dollars spent on this exploration's runs. Present only when cost
440
- * tracking was wired (`costOf`) — absent means "not tracked", never $0. */
441
- costUsd?: number;
442
- /** Runs whose cost was unknown (`costOf` returned null) — counted apart,
443
- * never folded into `costUsd` as a fabricated $0. */
444
- costUnknownRuns?: number;
445
- /** Evaluations that threw (transport/backend failures). They consumed no
446
- * run budget and scored nothing — an infra axis, never folded into
447
- * robustness or reported as findings. */
448
- evalErrors: number;
449
- /** Present when the run stopped before its budget because consecutive
450
- * eval errors tripped the circuit breaker (a dead backend must not burn
451
- * the remaining budget). The capsule-so-far is complete and honest. */
452
- stoppedEarly?: {
453
- reason: 'eval-errors';
454
- detail: string;
455
- };
145
+ target: string;
146
+ objective: string;
147
+ /** Stamped by the caller — the engine stays clock-free and deterministic. */
148
+ generatedAt?: string;
149
+ coverage: CoverageCell[];
150
+ /** Verified findings, sorted by descending interest. */
151
+ findings: Finding<S>[];
152
+ /** QD archive elites (binned by input × measured coords). */
153
+ archive: ArchiveEntry<S>[];
154
+ /** Post-harden lift, filled by a second pass after an improvement. */
155
+ lift?: {
156
+ before: number;
157
+ after: number;
158
+ verdict: string;
159
+ };
160
+ stats: {
161
+ totalRuns: number;
162
+ /** Input-cell denominator — the stratification plan. */
163
+ cellsTotal: number;
164
+ cellsCovered: number;
165
+ /** Distinct measured-descriptor bins observed (never part of the denominator). */
166
+ behaviorBinsObserved: number;
167
+ candidateFindings: number;
168
+ verifiedFindings: number;
169
+ /** Distribution of per-cell mean scores across covered cells (cells weigh
170
+ * equally — variance steering sends more runs to weak cells, so a
171
+ * run-weighted average would bias low). `null` when nothing ran. */
172
+ robustness: Distribution | null;
173
+ /** Evaluation wall-clock across all runs. `null` when nothing ran. */
174
+ latencyMs: Distribution | null;
175
+ /** Known dollars spent on this exploration's runs. Present only when cost
176
+ * tracking was wired (`costOf`) — absent means "not tracked", never $0. */
177
+ costUsd?: number;
178
+ /** Runs whose cost was unknown (`costOf` returned null) — counted apart,
179
+ * never folded into `costUsd` as a fabricated $0. */
180
+ costUnknownRuns?: number;
181
+ /** Evaluations that threw (transport/backend failures). They consumed no
182
+ * run budget and scored nothing — an infra axis, never folded into
183
+ * robustness or reported as findings. */
184
+ evalErrors: number;
185
+ /** Present when the run stopped before its budget because consecutive
186
+ * eval errors tripped the circuit breaker (a dead backend must not burn
187
+ * the remaining budget). The capsule-so-far is complete and honest. */
188
+ stoppedEarly?: {
189
+ reason: 'eval-errors';
190
+ detail: string;
456
191
  };
192
+ };
457
193
  }
458
194
  /**
459
195
  * Known cost of one evaluated run. `model` attributes the spend in the ledger's
@@ -461,121 +197,111 @@ interface CapsuleData<S> {
461
197
  * real either way — recorded as `actualCostUsd`, never an estimate).
462
198
  */
463
199
  interface RunCost {
464
- usd: number;
465
- model?: string;
200
+ usd: number;
201
+ model?: string;
466
202
  }
467
203
  type ExploreEvent<S> = {
468
- type: 'cell-allocated';
469
- cell: Cell;
470
- count: number;
204
+ type: 'cell-allocated';
205
+ cell: Cell;
206
+ count: number;
471
207
  } | {
472
- type: 'evaluated';
473
- cell: Cell;
474
- scenario: S;
475
- evaluation: Evaluation;
208
+ type: 'evaluated';
209
+ cell: Cell;
210
+ scenario: S;
211
+ evaluation: Evaluation;
476
212
  } | {
477
- type: 'finding';
478
- finding: Finding<S>;
213
+ type: 'finding';
214
+ finding: Finding<S>;
479
215
  } | {
480
- type: 'eval-error';
481
- cell: Cell;
482
- scenarioId: string;
483
- message: string;
216
+ type: 'eval-error';
217
+ cell: Cell;
218
+ scenarioId: string;
219
+ message: string;
484
220
  } | {
485
- type: 'round';
486
- runsUsed: number;
487
- budget: number;
221
+ type: 'round';
222
+ runsUsed: number;
223
+ budget: number;
488
224
  };
489
225
  interface ExploreOptions<S> {
490
- /** Name of the target under exploration — labels the capsule. */
491
- target: string;
492
- /** The input stratification plan. */
493
- space: BehaviorSpace;
494
- /** Candidate generator. */
495
- proposer: MutationProposer<S>;
496
- /** Runs the target → multi-objective `Evaluation`. */
497
- evaluate: Evaluator<S>;
498
- /** Seed corpus per cell. */
499
- seedsFor: (cell: Cell) => S[] | Promise<S[]>;
500
- /** Stable id for a scenario (dedup + lineage). */
501
- scenarioId: (scenario: S) => string;
502
- /** Human-legible text — drives capsule exemplars. */
503
- scenarioText?: (scenario: S) => string;
504
- /** Measured behavior coords appended to the archive bin. Default: input cell only. */
505
- descriptor?: (scenario: S, ev: Evaluation) => Record<string, string>;
506
- /** What "interesting" means. Default: `adversarialObjective(0.5)`. */
507
- objective?: Objective;
508
- /** Validity gates. Default pass-through. */
509
- gates?: ValidityGates<S>;
510
- /** Budget steering across input cells. `variance` chases uncertainty; `uniform` is the unsteered ablation baseline. Default `variance`. */
511
- allocation?: 'variance' | 'uniform';
512
- /** Total target evaluations. */
513
- budget: number;
514
- /** Minimum evaluations per input cell before steering. Default 2. */
515
- floorPerCell?: number;
516
- /** Shrink a notable scenario to its minimal trigger. Default: identity. */
517
- minimize?: (scenario: S, evaluate: Evaluator<S>, cell: Cell) => Promise<S> | S;
518
- /** Max concurrent `evaluate` calls. Default 1. */
519
- concurrency?: number;
520
- /** Stop the run after this many CONSECUTIVE eval errors (a dead backend must
521
- * not burn the remaining budget). Successes reset the streak. Default 5. */
522
- maxConsecutiveEvalErrors?: number;
523
- /** Cooperative cancellation. */
524
- signal?: AbortSignal;
525
- /** Progress stream. */
526
- onProgress?: (event: ExploreEvent<S>) => void;
527
- /** Deterministic seed. Default 1. */
528
- seed?: number;
529
- /**
530
- * Cost of one evaluated run — consumer-supplied; the explorer cannot know
531
- * token usage. Return null when the cost is unknown: the run is COUNTED in
532
- * `stats.costUnknownRuns`, never folded into the total as $0. Required by
533
- * every other cost option (`costBudgetUsd` / `ledger` / `onCost`).
534
- */
535
- costOf?: (scenario: S, cell: Cell, ev: Evaluation) => RunCost | null;
536
- /** Pre-call hard maximum. Required whenever the explorer uses a capped ledger. */
537
- maximumChargeOf?: (scenario: S, cell: Cell) => MaximumCharge;
538
- /**
539
- * Hard dollar cap. Each evaluation reserves `maximumChargeOf` before it starts;
540
- * calls that do not fit are rejected. Unknown totals stop further paid work.
541
- */
542
- costBudgetUsd?: number;
543
- /**
544
- * Sink for per-run cost entries — each known `costOf` result is recorded
545
- * with channel 'agent' and `actualCostUsd` (token axes are zero: the
546
- * explorer only sees dollars). Pass the program's shared `CostLedger` so
547
- * `costReport` stamps fuzz spend alongside judge/analyst spend.
548
- */
549
- ledger?: CostLedgerHandle;
550
- /** Observer fired for every known-cost run recorded. */
551
- onCost?: (entry: {
552
- usd: number;
553
- channel: CostChannel;
554
- }) => void;
226
+ /** Name of the target under exploration — labels the capsule. */
227
+ target: string;
228
+ /** The input stratification plan. */
229
+ space: BehaviorSpace;
230
+ /** Candidate generator. */
231
+ proposer: MutationProposer<S>;
232
+ /** Runs the target → multi-objective `Evaluation`. */
233
+ evaluate: Evaluator<S>;
234
+ /** Seed corpus per cell. */
235
+ seedsFor: (cell: Cell) => S[] | Promise<S[]>;
236
+ /** Stable id for a scenario (dedup + lineage). */
237
+ scenarioId: (scenario: S) => string;
238
+ /** Human-legible text — drives capsule exemplars. */
239
+ scenarioText?: (scenario: S) => string;
240
+ /** Measured behavior coords appended to the archive bin. Default: input cell only. */
241
+ descriptor?: (scenario: S, ev: Evaluation) => Record<string, string>;
242
+ /** What "interesting" means. Default: `adversarialObjective(0.5)`. */
243
+ objective?: Objective;
244
+ /** Validity gates. Default pass-through. */
245
+ gates?: ValidityGates<S>;
246
+ /** Budget steering across input cells. `variance` chases uncertainty; `uniform` is the unsteered ablation baseline. Default `variance`. */
247
+ allocation?: 'variance' | 'uniform';
248
+ /** Total target evaluations. */
249
+ budget: number;
250
+ /** Minimum evaluations per input cell before steering. Default 2. */
251
+ floorPerCell?: number;
252
+ /** Shrink a notable scenario to its minimal trigger. Default: identity. */
253
+ minimize?: (scenario: S, evaluate: Evaluator<S>, cell: Cell) => Promise<S> | S;
254
+ /** Max concurrent `evaluate` calls. Default 1. */
255
+ concurrency?: number;
256
+ /** Stop the run after this many CONSECUTIVE eval errors (a dead backend must
257
+ * not burn the remaining budget). Successes reset the streak. Default 5. */
258
+ maxConsecutiveEvalErrors?: number;
259
+ /** Cooperative cancellation. */
260
+ signal?: AbortSignal;
261
+ /** Progress stream. */
262
+ onProgress?: (event: ExploreEvent<S>) => void;
263
+ /** Deterministic seed. Default 1. */
264
+ seed?: number;
265
+ /**
266
+ * Cost of one evaluated run — consumer-supplied; the explorer cannot know
267
+ * token usage. Return null when the cost is unknown: the run is COUNTED in
268
+ * `stats.costUnknownRuns`, never folded into the total as $0. Required by
269
+ * every other cost option (`costBudgetUsd` / `ledger` / `onCost`).
270
+ */
271
+ costOf?: (scenario: S, cell: Cell, ev: Evaluation) => RunCost | null;
272
+ /** Pre-call hard maximum. Required whenever the explorer uses a capped ledger. */
273
+ maximumChargeOf?: (scenario: S, cell: Cell) => MaximumCharge;
274
+ /**
275
+ * Hard dollar cap. Each evaluation reserves `maximumChargeOf` before it starts;
276
+ * calls that do not fit are rejected. Unknown totals stop further paid work.
277
+ */
278
+ costBudgetUsd?: number;
279
+ /**
280
+ * Sink for per-run cost entries — each known `costOf` result is recorded
281
+ * with channel 'agent' and `actualCostUsd` (token axes are zero: the
282
+ * explorer only sees dollars). Pass the program's shared `CostLedger` so
283
+ * `costReport` stamps fuzz spend alongside judge/analyst spend.
284
+ */
285
+ ledger?: CostLedgerHandle;
286
+ /** Observer fired for every known-cost run recorded. */
287
+ onCost?: (entry: {
288
+ usd: number;
289
+ channel: CostChannel;
290
+ }) => void;
555
291
  }
556
-
557
- /**
558
- * Input-space tiling + coverage projection.
559
- *
560
- * Cells are the cartesian product of the input axes — the stratification plan,
561
- * enumerable up front so the planned-vs-covered denominator is honest. Coverage
562
- * is projected from the evaluation log: per cell, the full DISTRIBUTION of the
563
- * headline score, of each scored dimension, and of evaluation latency — a bare
564
- * mean hides outliers, so every aggregate carries its spread. Per-cell cost is
565
- * split known-dollars vs unknown-runs, never folded into a fabricated $0.
566
- */
567
-
292
+ //#endregion
293
+ //#region src/fuzz/cube.d.ts
568
294
  /** One recorded evaluation — the unit coverage and the capsule are built from. */
569
295
  interface EvalRecord {
570
- cell: Cell;
571
- ev: Evaluation;
572
- /** The objective's interest score for this evaluation. */
573
- interest: number;
574
- /** Evaluation wall-clock — engine-measured unless `ev.latencyMs` overrode it. */
575
- latencyMs: number;
576
- /** Known dollars for this run. `null` = cost tracking was wired but this
577
- * run's cost was unknowable (counted apart). Absent = not tracked at all. */
578
- costUsd?: number | null;
296
+ cell: Cell;
297
+ ev: Evaluation;
298
+ /** The objective's interest score for this evaluation. */
299
+ interest: number;
300
+ /** Evaluation wall-clock — engine-measured unless `ev.latencyMs` overrode it. */
301
+ latencyMs: number;
302
+ /** Known dollars for this run. `null` = cost tracking was wired but this
303
+ * run's cost was unknowable (counted apart). Absent = not tracked at all. */
304
+ costUsd?: number | null;
579
305
  }
580
306
  /** Enumerate every input cell (cartesian product of the axes), in stable order. */
581
307
  declare function enumerateCells(space: BehaviorSpace): Cell[];
@@ -586,132 +312,95 @@ declare function cellId(space: BehaviorSpace, coords: Record<string, string>): s
586
312
  * no evaluations reports `score: null` (honestly uncovered), never zeros.
587
313
  */
588
314
  declare function buildCoverage(cells: Cell[], log: EvalRecord[], threshold: number): CoverageCell[];
589
-
590
- /**
591
- * The capsule — the artifact every exploration produces.
592
- *
593
- * `buildCapsule` assembles coverage + verified findings + the QD archive into a
594
- * pure `CapsuleData` (no clock, no I/O — deterministic and snapshot-testable).
595
- * `renderCapsuleHtml` turns it into a standalone page: the input-cell heat-map
596
- * (planned vs covered), per-dimension weakness chips, and the minimized finding
597
- * exemplars. One artifact — the hardening map and the shareable proof object.
598
- */
599
-
315
+ //#endregion
316
+ //#region src/fuzz/capsule.d.ts
600
317
  interface BuildCapsuleInput<S> {
601
- target: string;
602
- objective: string;
603
- cells: Cell[];
604
- log: EvalRecord[];
605
- /** The objective's notable threshold — drives findingRate. */
606
- threshold: number;
607
- archive: ArchiveEntry<S>[];
608
- findings: Finding<S>[];
609
- candidateFindings: number;
610
- runsUsed: number;
611
- /** Known-dollar / unknown-run split — present only when cost tracking was
612
- * wired; the capsule never fabricates a $0 total. */
613
- cost?: {
614
- costUsd: number;
615
- costUnknownRuns: number;
616
- };
617
- /** Evaluations that threw — infra outcomes, never folded into robustness. */
618
- evalErrors: number;
619
- /** Set when the consecutive-error circuit breaker stopped the run early. */
620
- stoppedEarly?: {
621
- reason: 'eval-errors';
622
- detail: string;
623
- };
318
+ target: string;
319
+ objective: string;
320
+ cells: Cell[];
321
+ log: EvalRecord[];
322
+ /** The objective's notable threshold — drives findingRate. */
323
+ threshold: number;
324
+ archive: ArchiveEntry<S>[];
325
+ findings: Finding<S>[];
326
+ candidateFindings: number;
327
+ runsUsed: number;
328
+ /** Known-dollar / unknown-run split — present only when cost tracking was
329
+ * wired; the capsule never fabricates a $0 total. */
330
+ cost?: {
331
+ costUsd: number;
332
+ costUnknownRuns: number;
333
+ };
334
+ /** Evaluations that threw — infra outcomes, never folded into robustness. */
335
+ evalErrors: number;
336
+ /** Set when the consecutive-error circuit breaker stopped the run early. */
337
+ stoppedEarly?: {
338
+ reason: 'eval-errors';
339
+ detail: string;
340
+ };
624
341
  }
625
342
  declare function buildCapsule<S>(input: BuildCapsuleInput<S>): CapsuleData<S>;
626
343
  interface RenderCapsuleOptions {
627
- /** Max finding exemplars to show. Default 8. */
628
- maxFindings?: number;
629
- /** ISO timestamp to stamp into the page (keeps the pure capsule clock-free). */
630
- generatedAt?: string;
344
+ /** Max finding exemplars to show. Default 8. */
345
+ maxFindings?: number;
346
+ /** ISO timestamp to stamp into the page (keeps the pure capsule clock-free). */
347
+ generatedAt?: string;
631
348
  }
632
349
  /** Render a self-contained HTML capsule — heat-map + per-dimension chips + verified findings. */
633
350
  declare function renderCapsuleHtml<S>(capsule: CapsuleData<S>, opts?: RenderCapsuleOptions): string;
634
-
635
- /**
636
- * The exploration engine — a stateful session over a behavior space.
637
- *
638
- * Each `step()`: allocate budget across INPUT cells (floor first, then variance
639
- * steering toward the least-certain cells), propose candidates (the proposer
640
- * reads current elites + findings, so the search deepens generationally),
641
- * evaluate with bounded concurrency, archive the most interesting scenario per
642
- * input×measured bin, and admit notable candidates that pass the validity gates.
643
- * `run()` loops to budget. `coverage()`/`findings()`/`capsule()` read live state —
644
- * the surface `makeExploreTools` exposes so an agent can drive the session.
645
- *
646
- * One evaluation log (`EvalRecord[]`) is the source of truth; allocation
647
- * observations and coverage are projections of it.
648
- */
649
-
351
+ //#endregion
352
+ //#region src/fuzz/explorer.d.ts
650
353
  declare class BehaviorExplorer<S> {
651
- private readonly opts;
652
- private readonly cells;
653
- private readonly cellById;
654
- private readonly objective;
655
- private readonly threshold;
656
- private readonly floorPerCell;
657
- private readonly perRoundBudget;
658
- /** The single evaluation log — coverage + allocation are projections of it. */
659
- private readonly log;
660
- /** binId (input × measured coords) → the most interesting entry seen. */
661
- private readonly archiveByBin;
662
- private readonly _findings;
663
- private runsUsed;
664
- private candidateFindings;
665
- private evalErrors;
666
- private consecutiveEvalErrors;
667
- private stoppedEarly;
668
- private rngState;
669
- private readonly costLedger?;
670
- private readonly costPhase;
671
- constructor(opts: ExploreOptions<S>);
672
- private rng;
673
- private binId;
674
- private allocate;
675
- private objectiveContext;
676
- /** Elites whose INPUT cell matches — what the proposer mutates/deepens from. */
677
- private elitesFor;
678
- /** One allocate → propose → evaluate → gate → archive round. */
679
- step(): Promise<{
680
- runs: number;
681
- findings: Finding<S>[];
682
- }>;
683
- /** Loop `step()` until the run or dollar budget is spent, the signal aborts,
684
- * or no progress is made. */
685
- run(): Promise<CapsuleData<S>>;
686
- coverage(): CoverageCell[];
687
- findings(): Finding<S>[];
688
- capsule(): CapsuleData<S>;
689
- }
690
-
691
- /**
692
- * `fuzzAgent` — the adversarial batch preset over `BehaviorExplorer`.
693
- *
694
- * One call: explore the space to budget with the adversarial objective and
695
- * return the capsule. For agent-driven, incremental, or multi-objective use,
696
- * construct a `BehaviorExplorer` and drive it via `makeExploreTools`.
697
- */
698
-
354
+ private readonly opts;
355
+ private readonly cells;
356
+ private readonly cellById;
357
+ private readonly objective;
358
+ private readonly threshold;
359
+ private readonly floorPerCell;
360
+ private readonly perRoundBudget;
361
+ /** The single evaluation log — coverage + allocation are projections of it. */
362
+ private readonly log;
363
+ /** binId (input × measured coords) → the most interesting entry seen. */
364
+ private readonly archiveByBin;
365
+ private readonly _findings;
366
+ private runsUsed;
367
+ private candidateFindings;
368
+ private evalErrors;
369
+ private consecutiveEvalErrors;
370
+ private stoppedEarly;
371
+ private rngState;
372
+ private readonly costLedger?;
373
+ private readonly costPhase;
374
+ constructor(opts: ExploreOptions<S>);
375
+ private rng;
376
+ private binId;
377
+ private allocate;
378
+ private objectiveContext;
379
+ /** Elites whose INPUT cell matches — what the proposer mutates/deepens from. */
380
+ private elitesFor;
381
+ /** One allocate → propose → evaluate → gate → archive round. */
382
+ step(): Promise<{
383
+ runs: number;
384
+ findings: Finding<S>[];
385
+ }>;
386
+ /** Loop `step()` until the run or dollar budget is spent, the signal aborts,
387
+ * or no progress is made. */
388
+ run(): Promise<CapsuleData<S>>;
389
+ coverage(): CoverageCell[];
390
+ findings(): Finding<S>[];
391
+ capsule(): CapsuleData<S>;
392
+ }
393
+ //#endregion
394
+ //#region src/fuzz/fuzz-agent.d.ts
699
395
  type FuzzAgentOptions<S> = Omit<ExploreOptions<S>, 'objective'> & {
700
- /** Score strictly below this is a candidate failure. Default 0.5. */
701
- failureThreshold?: number;
396
+ /** Score strictly below this is a candidate failure. Default 0.5. */
397
+ failureThreshold?: number;
702
398
  };
703
399
  declare function fuzzAgent<S>(opts: FuzzAgentOptions<S>): Promise<{
704
- capsule: CapsuleData<S>;
400
+ capsule: CapsuleData<S>;
705
401
  }>;
706
-
707
- /**
708
- * Validity gates — what separates a fuzzer from a slop generator.
709
- *
710
- * A notable candidate is admitted only when it is fair and reproducible. None of
711
- * these are on by default: the live wiring opts in, so reported findings carry
712
- * their proof.
713
- */
714
-
402
+ //#endregion
403
+ //#region src/fuzz/gates.d.ts
715
404
  /** Combine gate sets; a candidate must pass every gate in every set. */
716
405
  declare function composeGates<S>(...sets: Array<ValidityGates<S> | undefined>): ValidityGates<S>;
717
406
  /**
@@ -722,38 +411,30 @@ declare function composeGates<S>(...sets: Array<ValidityGates<S> | undefined>):
722
411
  * evaluation per candidate (candidates are rare, so cheap).
723
412
  */
724
413
  declare function perturbationStabilityGate<S>(opts: {
725
- evaluate: Evaluator<S>;
726
- /** Produce a semantic-preserving rephrase. Return null to skip (treated as pass). */
727
- perturb: (scenario: S) => S | null;
728
- /** Score strictly below this still counts as failing. Default 0.5. */
729
- failureThreshold?: number;
414
+ evaluate: Evaluator<S>;
415
+ /** Produce a semantic-preserving rephrase. Return null to skip (treated as pass). */
416
+ perturb: (scenario: S) => S | null;
417
+ /** Score strictly below this still counts as failing. Default 0.5. */
418
+ failureThreshold?: number;
730
419
  }): ValidityGates<S>;
731
420
  /**
732
421
  * Severity-floor gate. Reject borderline candidates whose score sits in a band
733
422
  * just under the threshold — judge noise, not a real defect.
734
423
  */
735
424
  declare function severityFloorGate<S>(opts: {
736
- failureThreshold?: number;
737
- margin?: number;
425
+ failureThreshold?: number;
426
+ margin?: number;
738
427
  }): ValidityGates<S>;
739
-
740
- /**
741
- * Shipped policies for the exploration engine.
742
- *
743
- * `MutationProposer` is a plain function type — an agent running a generator skill
744
- * IS a proposer (`(ctx) => dispatchToSkill(ctx)`), no wrapper needed. `mutationProposer`
745
- * builds the deterministic, LLM-free one from mutation operators. Objectives are
746
- * interfaces because the engine reads `kind` + `threshold` off them.
747
- */
748
-
428
+ //#endregion
429
+ //#region src/fuzz/policies.d.ts
749
430
  /**
750
431
  * Perturbation-based search: apply the cell's mutation operators to the current
751
432
  * elites + seeds, deduping by id. Elites first — mutating the most interesting
752
433
  * scenario found so far is what makes the search deepen across rounds.
753
434
  */
754
435
  declare function mutationProposer<S>(opts: {
755
- mutationsFor: (cell: Cell) => AdversarialMutation<S>[];
756
- scenarioId: (s: S) => string;
436
+ mutationsFor: (cell: Cell) => AdversarialMutation<S>[];
437
+ scenarioId: (s: S) => string;
757
438
  }): MutationProposer<S>;
758
439
  /** Adversarial: a low headline score is interesting — find where the agent fails. */
759
440
  declare function adversarialObjective(threshold?: number): Objective;
@@ -763,25 +444,18 @@ declare function adversarialObjective(threshold?: number): Objective;
763
444
  * rather than re-finding the same hole.
764
445
  */
765
446
  declare function noveltyObjective(threshold?: number): Objective;
766
-
767
- /**
768
- * Agent-drivable surface over a live exploration session.
769
- *
770
- * Framework-neutral tool defs ({name, description, parameters: JSON Schema,
771
- * handler}) so the on-demand agent — not a batch script — drives the search:
772
- * step it, read coverage, inspect findings, render the capsule. Transport
773
- * encodings (OpenAI function shape, MCP) are one-line mappings the host owns.
774
- */
775
-
447
+ //#endregion
448
+ //#region src/fuzz/tools.d.ts
776
449
  interface ExploreToolDef {
777
- name: string;
778
- description: string;
779
- /** JSON Schema (draft-07+) for the arguments. */
780
- parameters: Record<string, unknown>;
781
- handler: (args: unknown, ctx?: {
782
- signal?: AbortSignal;
783
- }) => Promise<unknown>;
450
+ name: string;
451
+ description: string;
452
+ /** JSON Schema (draft-07+) for the arguments. */
453
+ parameters: Record<string, unknown>;
454
+ handler: (args: unknown, ctx?: {
455
+ signal?: AbortSignal;
456
+ }) => Promise<unknown>;
784
457
  }
785
458
  declare function makeExploreTools<S>(explorer: BehaviorExplorer<S>): ExploreToolDef[];
786
-
459
+ //#endregion
787
460
  export { type ArchiveEntry, BehaviorExplorer, type BehaviorSpace, type BuildCapsuleInput, type CapsuleData, type Cell, type CoverageCell, type EvalRecord, type Evaluation, type Evaluator, type ExploreEvent, type ExploreOptions, type ExploreToolDef, type Finding, type FuzzAgentOptions, type MutationProposer, type Objective, type ObjectiveContext, type ProposeContext, type RenderCapsuleOptions, type RunCost, type SpaceAxis, type ValidityGates, adversarialObjective, buildCapsule, buildCoverage, cellId, composeGates, enumerateCells, fuzzAgent, makeExploreTools, mutationProposer, noveltyObjective, perturbationStabilityGate, renderCapsuleHtml, severityFloorGate };
461
+ //# sourceMappingURL=fuzz.d.ts.map