@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,1739 @@
1
+ import { a as RunRecord } from "./run-record-CnZu_gjl.js";
2
+ import { s as TraceStore } from "./store-CT9YIIve.js";
3
+ import { c as CostLedgerHandle, f as CostLedgerSummary, o as CostLedger, p as CostReceipt, y as CustomTokenPricing } from "./cost-ledger-Dye6jCgg.js";
4
+ import { n as LlmCallMetadata } from "./llm-client-B_nIBlYo.js";
5
+ import { A as ChatClient } from "./types-DGsxbAEd.js";
6
+ import { f as PairedBootstrapResult } from "./statistics-Cmj6nynr.js";
7
+ import { C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, o as CampaignScenarioIdentity, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-k9tZGKUg.js";
8
+ import { a as DatasetScenario, t as Dataset } from "./dataset-BvtnC8Dc.js";
9
+ import { t as DetectRewardHackingInput } from "./reward-hacking-eAnOsynk.js";
10
+ import { g as TraceSpanEvent, t as HostedClient } from "./client-C97NMzqi.js";
11
+ import { z } from "zod";
12
+ //#region src/llm-judge.d.ts
13
+ /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
14
+ * bare string uses the key as its own description. */
15
+ type LlmJudgeDimension = string | JudgeDimension;
16
+ interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
17
+ /** The injected LLM transport. One `chat()` call per `score()`. Required —
18
+ * there is no default route, so a misconfigured judge fails at construction,
19
+ * never silently against the free-tier router. */
20
+ chat: ChatClient;
21
+ /** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
22
+ * returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
23
+ dimensions?: LlmJudgeDimension[];
24
+ /** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
25
+ model?: string;
26
+ /** Explicit scoring revision for opaque transport or renderer changes. */
27
+ judgeVersion?: string;
28
+ temperature?: number;
29
+ maxTokens?: number;
30
+ /** Composite weights forwarded to `weightedComposite`: a partial map selects
31
+ * AND weights exactly the named dimensions. Omit for a uniform mean. */
32
+ weights?: Record<string, number>;
33
+ /** Scale the model is prompted to score on, normalized into `[0,1]`:
34
+ * - `'unit'` (default): the model returns `[0,1]` directly.
35
+ * - `'ten'`: the model returns `[0,10]`; divided by 10 here.
36
+ * The prompt is annotated with the expected range either way. */
37
+ scale?: 'unit' | 'ten';
38
+ /** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
39
+ appliesTo?: (scenario: TScenario) => boolean;
40
+ /** Render the artifact + scenario into the user message. Default:
41
+ * pretty-printed JSON of `{ scenario, artifact }`. */
42
+ renderUser?: (input: {
43
+ artifact: TArtifact;
44
+ scenario: TScenario;
45
+ }) => string;
46
+ /** Strict runtime contract; its JSON Schema is sent to the provider. */
47
+ costLedger?: CostLedgerHandle;
48
+ responseSchema?: {
49
+ name: string;
50
+ schema: z.ZodObject;
51
+ };
52
+ }
53
+ /**
54
+ * Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
55
+ * against `prompt` and reduces the model's per-dimension scores to a canonical
56
+ * `JudgeScore` in `[0,1]`.
57
+ *
58
+ * The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
59
+ * "notes": "…" }`; the helper strips fenced JSON, validates every declared
60
+ * dimension is present and in range, normalizes by `scale`, and composites via
61
+ * `weightedComposite`.
62
+ */
63
+ declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
64
+ //#endregion
65
+ //#region src/reference-equivalence-judge.d.ts
66
+ declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
67
+ declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
68
+ readonly userRequest: 8000;
69
+ readonly expectedAnswer: 32000;
70
+ readonly candidateOutput: 32000;
71
+ };
72
+ interface ReferenceEquivalenceScenario extends Scenario {
73
+ userRequest: string;
74
+ expectedAnswer: string;
75
+ }
76
+ interface ReferenceEquivalenceJudgeInput {
77
+ userRequest: string;
78
+ expectedAnswer: string;
79
+ candidateOutput: string;
80
+ }
81
+ interface ReferenceEquivalenceJudgeOptions {
82
+ /** Injected transport. No implicit provider or credentials are selected. */
83
+ chat: ChatClient;
84
+ /** Falls back to the ChatClient's default model. */
85
+ model?: string;
86
+ /** Used only by the direct-call adapter. */
87
+ signal?: AbortSignal;
88
+ /** Optional receipt destination for direct calls; campaigns supply their own. */
89
+ costLedger?: CostLedgerHandle;
90
+ }
91
+ interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
92
+ kind: 'reference-equivalence';
93
+ version: string;
94
+ score: number;
95
+ rationale: string;
96
+ }
97
+ /** Build the campaign-native expected-answer judge. */
98
+ declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
99
+ /** Direct-call adapter over the campaign judge for product callers. */
100
+ declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
101
+ //#endregion
102
+ //#region src/campaign/auto-pr.d.ts
103
+ interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
104
+ /** Campaign result to attach to the PR. */
105
+ result: CampaignResult<TArtifact, TScenario>;
106
+ /** Gate verdict explaining the promotion. Substrate refuses to open a PR
107
+ * when `gate.decision !== 'ship'` — fails loud. */
108
+ gate: GateResult;
109
+ /** Promoted surface diff — typically the new system prompt addendum or
110
+ * full profile diff. Substrate writes it as the PR body. */
111
+ promotedDiff: string;
112
+ /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
113
+ ghOwner: string;
114
+ ghRepo: string;
115
+ /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
116
+ branch?: string;
117
+ /** PR title. Default includes manifest hash. */
118
+ title?: string;
119
+ /** Whether to actually open the PR or just dry-run. Default reads
120
+ * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
121
+ dryRun?: boolean;
122
+ /** Test seam — substitute `gh pr create` invocation. */
123
+ ghExec?: (args: string[]) => {
124
+ stdout: string;
125
+ stderr: string;
126
+ status: number;
127
+ };
128
+ }
129
+ interface OpenAutoPrResult {
130
+ opened: boolean;
131
+ prUrl?: string;
132
+ dryRun: boolean;
133
+ reason: string;
134
+ }
135
+ /**
136
+ * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
137
+ */
138
+ declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
139
+ //#endregion
140
+ //#region src/campaign/coverage.d.ts
141
+ /** Reject campaign designs whose denominator cannot be identified exactly. */
142
+ declare function assertCampaignDesign<TScenario extends Scenario>(scenarios: readonly TScenario[], reps: number): void;
143
+ /** Redacted but independently verifiable identity of one complete scenario. */
144
+ declare function campaignScenarioIdentity<TScenario extends Scenario>(scenario: TScenario): CampaignScenarioIdentity & Pick<TScenario, 'id' | 'kind'>;
145
+ /** Canonical split identity reconstructed from redacted scenario identities. */
146
+ declare function campaignSplitDigestFromIdentities(scenarios: readonly CampaignScenarioIdentity[], reps: number): `sha256:${string}`;
147
+ /** Canonical identity of the exact scenario payloads and replicate count. */
148
+ declare function campaignSplitDigest<TScenario extends Scenario>(scenarios: readonly TScenario[], reps: number): `sha256:${string}`;
149
+ /** Refuse a campaign whose retained task identities contradict its split digest. */
150
+ declare function assertCampaignSplitIdentity(scenarios: readonly CampaignScenarioIdentity[], reps: number, splitDigest: string): void;
151
+ //#endregion
152
+ //#region src/campaign/external-optimizer-contracts.d.ts
153
+ interface ExternalOptimizerRunnerCommand {
154
+ command?: string;
155
+ args?: readonly string[];
156
+ env?: NodeJS.ProcessEnv;
157
+ }
158
+ type ExternalOptimizerResumeMode = 'never' | 'if-compatible' | 'required';
159
+ type ExternalTextCandidate = string | Record<string, string>;
160
+ interface ExternalTextEvaluationRequest {
161
+ candidate: ExternalTextCandidate;
162
+ exampleId: string;
163
+ }
164
+ interface ExternalOptimizerModelBudget {
165
+ /** Maximum optimizer-model spend, independent of task-evaluation spend. */
166
+ maxCostUsd: number;
167
+ /** Network attempts, including provider retries. */
168
+ maxRequests: number;
169
+ /** Reject a request body above this byte count. */
170
+ maxRequestBytes: number;
171
+ /** Reject a provider response above this byte count. */
172
+ maxResponseBytes: number;
173
+ /** Reject a request asking the provider for more output tokens. */
174
+ maxOutputTokensPerRequest: number;
175
+ /** Rates used to estimate cost when the provider omits a valid `usage.cost`. */
176
+ pricing: CustomTokenPricing;
177
+ /** Per-provider-request deadline. Default: 300,000 ms. */
178
+ requestTimeoutMs?: number;
179
+ }
180
+ //#endregion
181
+ //#region src/campaign/storage.d.ts
182
+ /**
183
+ * `CampaignStorage` — the filesystem seam `runCampaign` writes through
184
+ * (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
185
+ *
186
+ * The default (`fsCampaignStorage`) is the Node filesystem — identical
187
+ * behavior to the inline `node:fs` calls it replaces, so existing CLI
188
+ * consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
189
+ * `Map`, so the substrate runs in environments WITHOUT a filesystem
190
+ * (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
191
+ * still produces its `CampaignResult` (cells + aggregates) in memory;
192
+ * artifacts/traces simply aren't persisted to disk.
193
+ *
194
+ * Paths are opaque keys to the in-memory adapter — it does not parse them,
195
+ * so the same `join(...)`-built paths work unchanged across both adapters.
196
+ */
197
+ interface CampaignStorage {
198
+ /** Ensure a directory exists (recursive). No-op for in-memory. */
199
+ ensureDir(dir: string): void;
200
+ /** Does this path exist (as a written file or an ensured dir)? */
201
+ exists(path: string): boolean;
202
+ /** Read a UTF-8 file; `undefined` when missing or unreadable. */
203
+ read(path: string): string | undefined;
204
+ /** Write a file (string or bytes). Parent dir is assumed ensured. */
205
+ write(path: string, content: string | Uint8Array): void;
206
+ /** Append only when the current UTF-8 byte length matches `expectedBytes`.
207
+ * Returns the new length, or undefined when another writer won. */
208
+ append(path: string, content: string, expectedBytes: number): number | undefined;
209
+ }
210
+ /** Node-filesystem storage — the default. Lazily requires `node:fs` so the
211
+ * module imports cleanly in non-Node runtimes (where the caller passes
212
+ * `inMemoryCampaignStorage` instead and never constructs this).
213
+ *
214
+ * `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
215
+ * `require` is a ReferenceError under `"type": "module"`, which is exactly
216
+ * the shape this package publishes. */
217
+ declare function fsCampaignStorage(): CampaignStorage;
218
+ /** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
219
+ * live in a `Map` for the duration of the run; the `CampaignResult` is
220
+ * fully populated, but nothing is persisted to disk. */
221
+ declare function inMemoryCampaignStorage(): CampaignStorage;
222
+ /** Open the durable spend account stored beside a logical run. */
223
+ declare function createRunCostLedger(input: {
224
+ storage: CampaignStorage;
225
+ runDir: string;
226
+ costCeilingUsd?: number;
227
+ }): CostLedger;
228
+ //#endregion
229
+ //#region src/campaign/run-campaign.d.ts
230
+ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
231
+ scenarios: TScenario[];
232
+ dispatch: DispatchFn<TScenario, TArtifact>;
233
+ /** Abort active dispatches when the owning operation is cancelled. */
234
+ signal?: AbortSignal;
235
+ /**
236
+ * Stable identity for the dispatch behavior, included in the manifest/cache
237
+ * key. Set this when the same function name can run different models,
238
+ * prompts, tools, or external config.
239
+ */
240
+ dispatchRef?: string;
241
+ judges?: JudgeConfig<TArtifact, TScenario>[];
242
+ /** Required for reproducibility. Default 42. */
243
+ seed?: number;
244
+ /** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
245
+ * bootstrap-tight intervals on critical eval. */
246
+ reps?: number;
247
+ /** When true (default), completed cells are cached by
248
+ * (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
249
+ resumable?: boolean;
250
+ /** Optional store — when present, every artifact + judge score is captured
251
+ * with the configured `captureSource`. Capture is default ON; pass `'off'`
252
+ * to disable. */
253
+ labeledStore?: LabeledScenarioStore | 'off';
254
+ captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
255
+ captureSourceVersionHash?: string;
256
+ /** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
257
+ costCeiling?: number;
258
+ /** Shared spend account. Improvement loops pass one ledger through every
259
+ * campaign so the ceiling and returned total are run-wide. */
260
+ costLedger?: CostLedgerHandle;
261
+ /** Attribution label for receipts recorded by this campaign. */
262
+ costPhase?: string;
263
+ /** Additional immutable receipt tags supplied by an owning workflow. */
264
+ costTags?: Readonly<Record<string, string>>;
265
+ /** Max concurrent cells. Default 2. */
266
+ maxConcurrency?: number;
267
+ /**
268
+ * Stop after the first dispatch or judge error. The failed cell is persisted
269
+ * before active sibling cells are aborted and drained, then the campaign
270
+ * rejects with the exact error thrown by that dispatch or judge.
271
+ * Default false preserves the normal behavior of returning failed cells and
272
+ * continuing the remaining schedule.
273
+ */
274
+ abortOnCellError?: boolean;
275
+ /**
276
+ * Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
277
+ * rejects within this window is a hang (a stalled model request, an
278
+ * exhausted runtime resource, a backend that never closes its stream). When
279
+ * set, the cell's `ctx.signal` is aborted. A dispatch that stops is recorded
280
+ * as an error (`dispatch exceeded <N>ms`). A dispatch that ignores
281
+ * cancellation rejects the campaign without publishing incomplete cost data.
282
+ * `undefined`/`0` means unbounded.
283
+ */
284
+ dispatchTimeoutMs?: number;
285
+ /**
286
+ * Time allowed for an aborted dispatch and its paid calls to stop before the
287
+ * campaign rejects without producing a result. Default 5 seconds.
288
+ */
289
+ dispatchShutdownTimeoutMs?: number;
290
+ /** Required: where artifacts + traces land. A bare name (not an absolute path)
291
+ * resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
292
+ * bundles never pollute a repo working tree. Pass an absolute path to override. */
293
+ runDir: string;
294
+ /** Subject repo for the shared run-dir root (defaults to the CWD basename).
295
+ * Only consulted when `runDir` is a bare name. */
296
+ repo?: string;
297
+ /** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
298
+ * at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
299
+ * refuses this when the caller wires `autoOnPromote !== 'none'`. */
300
+ tracing?: 'on' | 'off';
301
+ /**
302
+ * Per-cell usage expectation — the early, fine-grained sibling of the
303
+ * batch `assertRealBackend` guard. A cell that produced an artifact (no
304
+ * error) but reported `costUsd === 0` AND zero tokens is a stub: the
305
+ * dispatch never reported LLM activity via `ctx.cost`. Modes:
306
+ * - `'warn'` (default) — log the offending cell loudly, keep going.
307
+ * - `'assert'` — throw `BackendIntegrityError` on the first such cell
308
+ * (fail-fast; recommended for CI campaigns expecting real LLM calls).
309
+ * - `'off'` — no check (replay / deterministic-only / offline analysis).
310
+ */
311
+ expectUsage?: 'assert' | 'warn' | 'off';
312
+ /** Test seam — override the wall clock for deterministic tests. */
313
+ now?: () => Date;
314
+ /** Test seam — override per-cell trace writer factory. */
315
+ buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
316
+ /** Storage backend for run/cell dirs, the resumability cache, artifacts,
317
+ * and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
318
+ * Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
319
+ * (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
320
+ * produced; artifacts/traces just aren't persisted to disk. */
321
+ storage?: CampaignStorage;
322
+ /**
323
+ * Optional per-cell placement strategy. Returns an opaque string the
324
+ * substrate forwards as `ctx.placement` to the Dispatch — placement-aware
325
+ * Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
326
+ * each cell to the right worker, region, or sandbox. When unset, every
327
+ * cell receives `ctx.placement = undefined` and behaves identically to
328
+ * the in-process case.
329
+ *
330
+ * @example
331
+ * cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
332
+ */
333
+ cellPlacement?: (input: {
334
+ scenario: TScenario;
335
+ rep: number;
336
+ generation?: number;
337
+ }) => string | undefined;
338
+ }
339
+ /** Durable `<cell>/failure-receipt.json` written before a failed cell can
340
+ * trigger campaign-wide cancellation. The cell records dispatch measurements;
341
+ * `cost` covers every settled agent and judge call attributed to this exact run
342
+ * attempt. */
343
+ interface CampaignCellFailureReceipt<TArtifact = unknown> {
344
+ schemaVersion: 1;
345
+ runAttemptId: string;
346
+ recordedAt: string;
347
+ failure: {
348
+ stage: 'dispatch' | 'judge';
349
+ judge?: string;
350
+ error: {
351
+ name: string;
352
+ message: string;
353
+ stack?: string;
354
+ };
355
+ };
356
+ cell: CampaignCellResult<TArtifact>;
357
+ cost: CostLedgerSummary;
358
+ }
359
+ /**
360
+ * Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
361
+ */
362
+ declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
363
+ interface CampaignRunPlanCell {
364
+ cellId: string;
365
+ scenarioId: string;
366
+ rep: number;
367
+ seed: number;
368
+ cachePath: string;
369
+ status: 'cached' | 'run';
370
+ reason?: 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'corrupt' | 'resumable-off';
371
+ }
372
+ interface CampaignRunPlan {
373
+ manifestHash: string;
374
+ splitDigest: `sha256:${string}`;
375
+ totalCells: number;
376
+ cellsCached: number;
377
+ cellsToRun: number;
378
+ cells: CampaignRunPlanCell[];
379
+ }
380
+ interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
381
+ scenarios: TScenario[];
382
+ dispatch?: DispatchFn<TScenario, TArtifact>;
383
+ dispatchRef?: string;
384
+ judges?: JudgeConfig<TArtifact, TScenario>[];
385
+ seed?: number;
386
+ reps?: number;
387
+ resumable?: boolean;
388
+ runDir: string;
389
+ /** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
390
+ repo?: string;
391
+ storage?: CampaignStorage;
392
+ }
393
+ /**
394
+ * Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
395
+ * run-vs-cached schedule so callers can preview cost and resumability before spending.
396
+ */
397
+ declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
398
+ //#endregion
399
+ //#region src/campaign/presets/compare-optimization-methods.d.ts
400
+ /** Shared campaign settings applied to every optimization method. */
401
+ type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
402
+ /** Cost reported by a method or by final test scoring. */
403
+ interface ComparisonCost {
404
+ totalCostUsd: number;
405
+ accountingComplete: boolean;
406
+ incompleteReasons: string[];
407
+ }
408
+ interface OptimizationPackageSource {
409
+ kind: 'package';
410
+ /** Whether package identity was inspected or supplied by caller code. */
411
+ evidence: 'observed' | 'declared';
412
+ package: string;
413
+ version: string;
414
+ sourceUrl?: string;
415
+ revision?: string;
416
+ /** SHA-256 of all installed module files observed before the run. */
417
+ sourceSha256?: string;
418
+ }
419
+ interface OptimizationModuleSource {
420
+ module: string;
421
+ sourceSha256: string;
422
+ }
423
+ interface OptimizationPythonRuntime {
424
+ implementation: string;
425
+ version: string;
426
+ }
427
+ interface OptimizationTokenUsage {
428
+ /** All input tokens, including cache reads and cache creation. */
429
+ inputTokens: number;
430
+ /** Input tokens served from a provider cache. */
431
+ cachedInputTokens?: number;
432
+ /** Input tokens used to create or write a provider cache entry. */
433
+ cacheWriteInputTokens?: number;
434
+ outputTokens: number;
435
+ /** Reasoning tokens included in `outputTokens`. */
436
+ reasoningTokens?: number;
437
+ totalTokens: number;
438
+ calls: number;
439
+ }
440
+ interface OptimizationMethodProvenance {
441
+ /** External optimizer package. */
442
+ source: OptimizationPackageSource;
443
+ /** Python bridge package that invoked the optimizer. */
444
+ bridge?: OptimizationPackageSource;
445
+ /** Custom engine modules imported by the optimizer. */
446
+ modules?: OptimizationModuleSource[];
447
+ /** Python implementation used by the bridge process. */
448
+ python?: OptimizationPythonRuntime;
449
+ /** Exact model identifier configured for optimizer-owned model calls. */
450
+ optimizerModel?: string;
451
+ runId: string;
452
+ /** Content identity shared by compatible resumptions. */
453
+ compatibleRunId?: string;
454
+ resumed: boolean;
455
+ evaluationCount: number;
456
+ artifactDir: string;
457
+ tokenUsage?: OptimizationTokenUsage;
458
+ }
459
+ /** Shared inputs for one optimization method. Final test data is absent. */
460
+ interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
461
+ /** Surface every method starts from. */
462
+ readonly baselineSurface: MutableSurface;
463
+ /** Evidence used to author or fit candidates. */
464
+ readonly trainScenarios: readonly TScenario[];
465
+ /** Data used for candidate acceptance, early stopping, and model selection. */
466
+ readonly selectionScenarios: readonly TScenario[];
467
+ /** Runs one scenario with a candidate surface. */
468
+ readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
469
+ /** Scores artifacts produced by `dispatchWithSurface`. */
470
+ readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
471
+ /** Method-specific artifacts are written below this directory. */
472
+ readonly runDir: string;
473
+ readonly seed: number;
474
+ /** Shared defaults for every method. A method may override them explicitly. */
475
+ readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
476
+ /** Durable spend account shared by every method and final scoring. */
477
+ readonly costLedger: CostLedgerHandle;
478
+ }
479
+ interface OptimizationMethodResult {
480
+ /** Surface selected without using the final test partition. */
481
+ winnerSurface: MutableSurface;
482
+ /** Optimization spend. Excludes final test scoring. */
483
+ cost: ComparisonCost;
484
+ /** Optimization duration. Excludes final test scoring. */
485
+ durationMs?: number;
486
+ /** Exact external implementation and run identity, when the method uses one. */
487
+ provenance?: OptimizationMethodProvenance;
488
+ }
489
+ /** A complete optimization method, including candidate generation and selection. */
490
+ interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
491
+ /** Unique, trimmed display name. Its normalized form must also be unique. */
492
+ name: string;
493
+ optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
494
+ }
495
+ interface OptimizationMethodScore {
496
+ name: string;
497
+ /** Mean final-test composite of the baseline (identical across methods). */
498
+ baselineComposite: number;
499
+ /** Mean final-test composite of this method's selected surface. */
500
+ winnerComposite: number;
501
+ /** Mean per-scenario final-test lift (winner minus baseline). */
502
+ lift: number;
503
+ /** Simultaneous paired-bootstrap interval for per-scenario lift.
504
+ * `low > 0` excludes zero after adjustment for all reported contrasts. */
505
+ liftCi: {
506
+ low: number;
507
+ high: number;
508
+ };
509
+ /** Optimization spend reported by the method. Excludes final test scoring. */
510
+ optimizationCost: ComparisonCost;
511
+ /** Optimization duration reported by the method. Excludes final test scoring. */
512
+ durationMs?: number;
513
+ /** Exact external implementation and run identity, when reported by the method. */
514
+ provenance?: OptimizationMethodProvenance;
515
+ /** Paired final-test values used to compute lift and its interval. */
516
+ scenarioScores: Array<{
517
+ scenarioId: string;
518
+ baselineComposite: number;
519
+ winnerComposite: number;
520
+ lift: number;
521
+ }>;
522
+ winnerSurface: MutableSurface;
523
+ /** 1-based, by descending lift. */
524
+ rank: number;
525
+ }
526
+ interface OptimizationMethodPairwise {
527
+ /** Higher-ranked method. */
528
+ a: string;
529
+ b: string;
530
+ /** Mean per-scenario untouched-test delta (a − b). */
531
+ deltaMean: number;
532
+ low: number;
533
+ high: number;
534
+ /** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
535
+ favored: string;
536
+ }
537
+ interface OptimizationMethodComparison {
538
+ /** Sorted by descending lift; `rank` set accordingly. */
539
+ scores: OptimizationMethodScore[];
540
+ best: OptimizationMethodScore;
541
+ /** Best vs each other method, using simultaneous paired-bootstrap intervals. */
542
+ pairwise: OptimizationMethodPairwise[];
543
+ testScenarioIds: string[];
544
+ /** Sum of the costs reported by every optimization method. */
545
+ optimizationCost: ComparisonCost;
546
+ /** Baseline and distinct winner scoring on the final test partition. */
547
+ testCost: ComparisonCost;
548
+ /** Optimization plus final test scoring. */
549
+ totalCost: ComparisonCost;
550
+ /** Caller-requested simultaneous coverage across all reported contrasts. */
551
+ confidence: number;
552
+ /** Bonferroni-adjusted confidence used for each bootstrap interval. */
553
+ intervalConfidence: number;
554
+ /** Method-vs-baseline plus all possible method-vs-method contrasts. */
555
+ comparisonCount: number;
556
+ /** Deterministic bootstrap and campaign seed. */
557
+ seed: number;
558
+ /** Bootstrap draws used for each interval. */
559
+ resamples: number;
560
+ /** Agent runs averaged within each test scenario before resampling scenarios. */
561
+ reps: number;
562
+ }
563
+ interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
564
+ methods: OptimizationMethod<TScenario, TArtifact>[];
565
+ baselineSurface: MutableSurface;
566
+ /** Evidence used by every optimizer to author or fit candidates. */
567
+ trainScenarios: TScenario[];
568
+ /** Candidate acceptance, early-stopping, and optimizer-selection data. */
569
+ selectionScenarios: TScenario[];
570
+ /** Untouched final comparison data. Never passed to an optimization method. */
571
+ testScenarios: TScenario[];
572
+ /** Scores a surface on a scenario. The methods and final test share this function. */
573
+ dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
574
+ judges: JudgeConfig<TArtifact, TScenario>[];
575
+ /** Bootstrap resamples for the lift intervals. Default is at least 2000 and
576
+ * rises when the requested simultaneous confidence needs finer tails. */
577
+ resamples?: number;
578
+ /** Shared defaults for each method's train and selection campaigns. */
579
+ optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
580
+ /** Number of optimization methods to run concurrently. Default 1. */
581
+ optimizationConcurrency?: number;
582
+ /** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
583
+ * Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
584
+ confidence?: number;
585
+ /** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
586
+ costCeiling?: number;
587
+ }
588
+ /**
589
+ * Compare complete optimization methods on disjoint train, selection, and final test data.
590
+ */
591
+ declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
592
+ /** Keep the cost fields a custom optimization method must report. */
593
+ declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
594
+ /** Preserve every optimizer token class while keeping total input and output explicit. */
595
+ declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
596
+ //#endregion
597
+ //#region src/campaign/external-text-evaluation.d.ts
598
+ interface ExternalOptimizationExample {
599
+ id: string;
600
+ data: unknown;
601
+ }
602
+ interface ExternalTextEvaluationResponse {
603
+ score: number;
604
+ info: {
605
+ scenarioId: string;
606
+ dimensions: Record<string, number>;
607
+ notes?: string;
608
+ artifact?: unknown;
609
+ };
610
+ }
611
+ //#endregion
612
+ //#region src/campaign/external-text-optimization-contract.d.ts
613
+ interface ExternalTextOptimizerContext {
614
+ readonly runId: string;
615
+ readonly name: string;
616
+ readonly objective: string;
617
+ readonly evaluationId: string;
618
+ readonly background?: string;
619
+ readonly seedCandidate: ExternalTextCandidate;
620
+ readonly trainSet: readonly ExternalOptimizationExample[];
621
+ readonly selectionSet: readonly ExternalOptimizationExample[];
622
+ readonly maxEvaluations: number;
623
+ readonly seed: number;
624
+ /** Stable directory for optimizer checkpoints from compatible attempts. */
625
+ readonly stateDir: string;
626
+ readonly restoreRequested: boolean;
627
+ readonly artifactDir: string;
628
+ readonly signal: AbortSignal;
629
+ /** Record every optimizer-owned paid call through this attributed account. */
630
+ readonly cost: CampaignCostMeter;
631
+ readonly evaluate: (request: ExternalTextEvaluationRequest) => Promise<ExternalTextEvaluationResponse>;
632
+ }
633
+ interface ExternalTextOptimizerResult {
634
+ bestCandidate: ExternalTextCandidate;
635
+ resumed: boolean;
636
+ costAccounting: {
637
+ kind: 'metered';
638
+ } | {
639
+ kind: 'no-paid-work';
640
+ } | {
641
+ kind: 'external';
642
+ reason: string;
643
+ };
644
+ }
645
+ /**
646
+ * Configuration for adapting another text optimizer.
647
+ *
648
+ * `run` owns search. Agent Eval owns split isolation, bounded candidate
649
+ * evaluation, exact cost collection, provenance, and final comparison.
650
+ */
651
+ interface ExternalTextOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
652
+ name: string;
653
+ source: Omit<OptimizationPackageSource, 'evidence'>;
654
+ objective: string;
655
+ evaluationId: string;
656
+ background?: string;
657
+ maxEvaluations: number;
658
+ /** Hard limit for calls made through `context.cost`. Use 0 for no paid work. */
659
+ maxOptimizerCostUsd: number;
660
+ /** Abort `context.signal` after this duration. Default: 30 minutes. */
661
+ timeoutMs?: number;
662
+ /** Default: `never`. Compatible runs reuse one state directory. */
663
+ resume?: ExternalOptimizerResumeMode;
664
+ maxCandidateChars?: number;
665
+ maxEvidenceChars?: number;
666
+ describeScenario?: (scenario: TScenario) => unknown;
667
+ describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
668
+ run: (context: ExternalTextOptimizerContext) => Promise<ExternalTextOptimizerResult>;
669
+ }
670
+ //#endregion
671
+ //#region src/campaign/external-text-optimization.d.ts
672
+ /**
673
+ * Adapt a third-party text optimizer without reimplementing its search.
674
+ *
675
+ * The callback never receives final test cases. Calls to `evaluate` are
676
+ * counted before execution and stop at `maxEvaluations`.
677
+ */
678
+ declare function externalTextOptimizationMethod<TScenario extends Scenario, TArtifact>(config: ExternalTextOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
679
+ //#endregion
680
+ //#region src/campaign/gates/compose.d.ts
681
+ /** Compose gates — all must `ship` for the composite to `ship`. First
682
+ * non-ship verdict short-circuits the composite verdict, but ALL gates run
683
+ * (so the result records every gate's reason — useful for diagnostics). */
684
+ declare function composeGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(...gates: Array<Gate<TArtifact, TScenario>>): Gate<TArtifact, TScenario>;
685
+ //#endregion
686
+ //#region src/canary.d.ts
687
+ type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
688
+ type CanarySeverity = 'info' | 'warn' | 'error';
689
+ interface CanaryAlert {
690
+ kind: CanaryKind;
691
+ severity: CanarySeverity;
692
+ message: string;
693
+ /** Numbers that informed the decision — drop straight into a
694
+ * dashboard / paper figure. */
695
+ evidence: Record<string, unknown>;
696
+ }
697
+ interface CanaryReport {
698
+ alerts: CanaryAlert[];
699
+ /** Per-kind summary count. */
700
+ counts: Record<CanaryKind, number>;
701
+ /** Whether each enabled detector had enough observations to run. */
702
+ evaluations: CanaryEvaluation[];
703
+ }
704
+ interface CanaryEvaluation {
705
+ kind: CanaryKind;
706
+ status: 'evaluated' | 'not_evaluated';
707
+ observations: number;
708
+ reason?: string;
709
+ }
710
+ interface CanaryOptions {
711
+ /**
712
+ * Silent-fallback detection.
713
+ * - `constant`: confidence value treated as the fallback signal.
714
+ * Default 0.30 (matches the soft-fail default in
715
+ * `propose-review.ts`).
716
+ * - `consecutiveThreshold`: trip the alert after this many
717
+ * consecutive runs at `constant` (or `fallback === true`).
718
+ * Default 3.
719
+ */
720
+ silentFallback?: {
721
+ constant?: number;
722
+ consecutiveThreshold?: number;
723
+ /** Floating-point tolerance when comparing against `constant`. */
724
+ epsilon?: number;
725
+ };
726
+ /**
727
+ * Calibration-drift detection.
728
+ * - `historyWindow`: number of past runs (oldest-first) treated as
729
+ * the historical baseline. Default 50.
730
+ * - `recentWindow`: number of recent runs (newest-first) compared
731
+ * against history. Default 20.
732
+ * - `ksAlpha`: alpha for the KS statistic vs critical value.
733
+ * Default 0.05.
734
+ * - `minRecent`: minimum recent runs required to even attempt the
735
+ * check. Default 10.
736
+ */
737
+ calibrationDrift?: {
738
+ historyWindow?: number;
739
+ recentWindow?: number;
740
+ ksAlpha?: number;
741
+ minRecent?: number;
742
+ };
743
+ /**
744
+ * Distribution-shift detection.
745
+ * - `category`: function that maps a run to a categorical bucket.
746
+ * Required to enable this canary; if omitted the chi-square check
747
+ * is skipped entirely.
748
+ * - `chiSquareAlpha`: alpha. Default 0.05.
749
+ * - `historyWindow`, `recentWindow`, `minRecent`: like above.
750
+ */
751
+ distributionShift?: {
752
+ category: (run: RunRecord) => string | null;
753
+ chiSquareAlpha?: number;
754
+ historyWindow?: number;
755
+ recentWindow?: number;
756
+ minRecent?: number;
757
+ };
758
+ }
759
+ /**
760
+ * Run all configured canaries against a chronological run list.
761
+ * Runs MUST be sorted oldest-to-newest by the caller — the order of
762
+ * the input is used to define "recent" vs "historical" windows.
763
+ */
764
+ declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
765
+ //#endregion
766
+ //#region src/red-team.d.ts
767
+ type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
768
+ interface RedTeamPayload {
769
+ category: RedTeamCategory;
770
+ /** The hostile input the agent sees. */
771
+ input: string;
772
+ /**
773
+ * What a safe agent should do: "refuse" means explicit refusal,
774
+ * "ignore" means process benignly without executing the attack.
775
+ */
776
+ expected: 'refuse' | 'ignore';
777
+ /** Substring(s) that MUST NOT appear in output. */
778
+ forbiddenStrings?: string[];
779
+ /** Tool names the agent MUST NOT call for this scenario. */
780
+ forbiddenTools?: string[];
781
+ }
782
+ interface RedTeamCase extends DatasetScenario {
783
+ payload: RedTeamPayload;
784
+ }
785
+ interface RedTeamFinding {
786
+ scenarioId: string;
787
+ category: RedTeamCategory;
788
+ passed: boolean;
789
+ reason: string;
790
+ evidence?: string;
791
+ }
792
+ interface RedTeamReport {
793
+ findings: RedTeamFinding[];
794
+ passRateByCategory: Record<RedTeamCategory, number>;
795
+ overallPassRate: number;
796
+ }
797
+ /** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
798
+ declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
799
+ declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
800
+ /**
801
+ * Score a single agent response against the red-team case.
802
+ * `toolCalls` should be the names of tools the agent invoked during the run.
803
+ */
804
+ declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
805
+ /** Aggregate red-team findings into per-category pass rates. */
806
+ declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
807
+ /**
808
+ * Extract the tool-call names from a corpus run — convenience for the
809
+ * common pipeline (run the scenario → score the run).
810
+ */
811
+ declare function toolNamesForRun(store: TraceStore, runId: string): Promise<string[]>;
812
+ //#endregion
813
+ //#region src/campaign/gates/default-production-gate.d.ts
814
+ type DefaultProductionGateCheck = 'dimension-regression' | 'budget' | 'red-team' | 'reward-hacking' | 'canary';
815
+ type DefaultProductionRewardHackingOptions = Omit<DetectRewardHackingInput, 'runs' | 'truthOf'> & {
816
+ truthOf: NonNullable<DetectRewardHackingInput['truthOf']>;
817
+ };
818
+ interface DefaultProductionGateOptions {
819
+ /** Required: scenarios held out from training; substrate compares
820
+ * candidate-on-holdout vs baseline-on-holdout. */
821
+ holdoutScenarios: Scenario[];
822
+ /** Minimum held-out lift the **paired-bootstrap CI lower bound** must clear
823
+ * to ship — NOT a point estimate. Default 0 ⇒ "confidently positive at the
824
+ * confidence level". Interpreted in the judge's native composite scale (set
825
+ * e.g. 2 for a 0-100 rubric to require a ≥2-point significant gain). */
826
+ deltaThreshold?: number;
827
+ /** Confidence level for the held-out + dimension bootstraps. Default 0.95. */
828
+ confidence?: number;
829
+ /** Bootstrap resamples. Default 2000. */
830
+ bootstrapResamples?: number;
831
+ /** Fixed bootstrap seed for a deterministic verdict. Default 1337. */
832
+ bootstrapSeed?: number;
833
+ /** Minimum paired holdout observations (scenarios × reps) before a
834
+ * significance claim is allowed; below it the gate HOLDS with `few_runs`
835
+ * rather than reading a degenerate CI. Default 3. */
836
+ minProductiveRuns?: number;
837
+ /** Ship statistic for the held-out significance test. Default `'mean'`
838
+ * (tie-robust — see `heldoutSignificance`). Pass `'median'` for
839
+ * outlier-robustness at the cost of tie-blindness. */
840
+ heldoutStatistic?: 'mean' | 'median';
841
+ /** Critical judge dimensions that must NOT significantly regress even when
842
+ * the net composite rises (anti-Goodhart). The gate HOLDS if any listed
843
+ * dimension's paired-delta CI lower bound < −`regressionTolerance`. E.g.
844
+ * `['hallucination_free']` for a legal agent. */
845
+ criticalDimensions?: string[];
846
+ /** Tolerance for the per-dimension regression guard, in the dimension's
847
+ * native scale. When omitted it auto-scales off observed magnitudes:
848
+ * 0.05 on [0,1], 5 on 0-100. */
849
+ regressionTolerance?: number;
850
+ /** Total $ budget for the complete improvement run. Requires
851
+ * `GateContext.costLedger`; missing or incomplete accounting holds. */
852
+ budgetUsd?: number;
853
+ /** Static artifact-screening cases. Only `expected: 'ignore'` cases without
854
+ * tool assertions are valid because this check does not dispatch case inputs
855
+ * or observe tool calls. */
856
+ redTeamBattery?: RedTeamCase[];
857
+ /** Shared run history, oldest first. Supplying history does not enable either
858
+ * monitoring check; configure `rewardHacking` and/or `canary` explicitly. */
859
+ recentRuns?: RunRecord[];
860
+ /** Enable reward-hacking monitoring with a caller-owned independent truth channel. */
861
+ rewardHacking?: DefaultProductionRewardHackingOptions;
862
+ /** Enable canary monitoring. Pass `{}` to use the canary defaults. */
863
+ canary?: CanaryOptions;
864
+ /** Optional checks that must be evaluated even when their normal input is
865
+ * absent. Configuring a check's input also makes that check required.
866
+ * Missing evidence always records `not_evaluated`; required unevaluated
867
+ * checks hold the release decision. Held-out significance is always required. */
868
+ requiredChecks?: DefaultProductionGateCheck[];
869
+ }
870
+ /**
871
+ * Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
872
+ */
873
+ declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
874
+ //#endregion
875
+ //#region src/campaign/gates/heldout-gate.d.ts
876
+ interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
877
+ scenarios: TScenario[];
878
+ /** Effect-size threshold the CI lower bound must clear, in the judge's native
879
+ * scale. Default 0.5. Equality holds; CI.low must be greater than this value. */
880
+ deltaThreshold?: number;
881
+ /** Bootstrap CI confidence. Default 0.95. */
882
+ confidence?: number;
883
+ /** Minimum paired holdout observations to claim significance. Default 3. */
884
+ minProductiveRuns?: number;
885
+ /** Bootstrap resamples. Default 2000. */
886
+ resamples?: number;
887
+ /** Fixed bootstrap seed for deterministic verdicts. Default 1337. */
888
+ bootstrapSeed?: number;
889
+ }
890
+ /**
891
+ * Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
892
+ * of the candidate-minus-baseline composite delta clears `deltaThreshold`.
893
+ */
894
+ declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
895
+ //#endregion
896
+ //#region src/campaign/gates/power-preflight.d.ts
897
+ /**
898
+ * Power preflight — "can this budget detect the effect you are hunting?"
899
+ *
900
+ * The failure it prevents (measured, twice): a live prompt-improvement campaign ran
901
+ * 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
902
+ * (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
903
+ * that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
904
+ * any effect a prompt change plausibly produces. The budget was spent learning what
905
+ * a 30-second calculation on the baseline cells already knew. No eval framework we
906
+ * know of surfaces this; every underpowered improvement run everywhere ends in an
907
+ * uninformative "hold".
908
+ *
909
+ * Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
910
+ * bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
911
+ * true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
912
+ * before the candidate exists; we bound it by the zero-correlation case
913
+ * `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
914
+ * direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
915
+ *
916
+ * Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
917
+ * live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
918
+ * it to every result and warns when the run was structurally unable to ship.
919
+ */
920
+ interface PowerPreflightOptions {
921
+ /** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
922
+ baselineComposites: number[];
923
+ /** Paired observations the budgeted comparison will produce
924
+ * (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
925
+ pairedN?: number;
926
+ /** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
927
+ deltaThreshold?: number;
928
+ /** CI confidence the gate uses. Default 0.95. */
929
+ confidence?: number;
930
+ /** True when the holdout is scored by the SAME judge/scorer family as the gate
931
+ * (selfImprove's default composition — one judge scores everything). Under a
932
+ * shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
933
+ * systematic judge bias is untouched, so the MDE here is a lower bound and the
934
+ * only full debiaser is an independent second scoring channel
935
+ * (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
936
+ sharedScorerChannel?: boolean;
937
+ }
938
+ interface PowerPreflight {
939
+ /** Paired observations the comparison will have. */
940
+ n: number;
941
+ /** Baseline per-cell composite standard deviation (the variance the effect must beat). */
942
+ sd: number;
943
+ /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
944
+ mde: number;
945
+ /** Baseline holdout composite mean. */
946
+ baselineMean: number;
947
+ /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
948
+ headroom: number;
949
+ /** True when even the largest achievable effect (headroom) is below the MDE —
950
+ * the run is structurally unable to ship regardless of proposal quality.
951
+ * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
952
+ underpowered: boolean;
953
+ /** True when composites look [0,1]-scaled; headroom/underpowered are only
954
+ * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
955
+ scaleAssumed: boolean;
956
+ deltaThreshold: number;
957
+ confidence: number;
958
+ /** Set when the holdout shares the gate's scoring channel: more cells cannot
959
+ * buy back systematic judge bias — treat the MDE as a lower bound. */
960
+ sharedChannelCaveat?: string;
961
+ /** One actionable sentence for humans and logs. */
962
+ recommendation: string;
963
+ }
964
+ /** Estimate the minimum detectable lift a paired-holdout improvement run can
965
+ * ship at a given budget, from the baseline holdout composites — call it BEFORE
966
+ * spending a search to learn whether the effect you are hunting is even
967
+ * observable at this holdout size and worker variance. */
968
+ declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
969
+ //#endregion
970
+ //#region src/pareto.d.ts
971
+ /**
972
+ * Pareto frontier — multi-objective optimization over candidate runs.
973
+ *
974
+ * Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
975
+ * trading off (cost, latency, quality) or (passRate, tokenBudget,
976
+ * ttfb), you rarely have a single "winner" — you have a set of
977
+ * non-dominated candidates. This module exposes:
978
+ *
979
+ * - `paretoFrontier`: filter a set of candidates to the non-dominated ones
980
+ * - `dominates`: does A dominate B across all objectives?
981
+ *
982
+ * Each objective is declared with a direction: 'maximize' (higher=better)
983
+ * or 'minimize' (lower=better). Candidates are any object; pass an
984
+ * `objective(candidate)` accessor.
985
+ */
986
+ type Direction = 'maximize' | 'minimize';
987
+ interface Objective<T> {
988
+ /** Stable label used in reports. */
989
+ name: string;
990
+ direction: Direction;
991
+ value: (candidate: T) => number;
992
+ }
993
+ interface ParetoResult<T> {
994
+ frontier: T[];
995
+ dominated: T[];
996
+ /** Index map: frontier[i] dominates each of dominatedBy[i]. */
997
+ dominanceMap: Array<{
998
+ dominator: T;
999
+ dominated: T[];
1000
+ }>;
1001
+ }
1002
+ /** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
1003
+ declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
1004
+ /**
1005
+ * Compute the non-dominated frontier. Candidates with NaN/Infinity on any
1006
+ * objective are excluded (can't rank them). A candidate enters the frontier
1007
+ * iff no other candidate dominates it.
1008
+ */
1009
+ declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
1010
+ /**
1011
+ * Weighted-sum scalarisation. Use as a tie-break / single-winner selector
1012
+ * when callers don't want to consume a frontier. Each objective contributes
1013
+ * its normalised value (0..1 via min-max across the candidate pool) times
1014
+ * its weight; missing weights default to 1/N.
1015
+ *
1016
+ * Direction is honoured automatically — `minimize` axes have their values
1017
+ * inverted before scaling so "higher scalar = better" always holds.
1018
+ */
1019
+ declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
1020
+ weights?: Partial<Record<string, number>>;
1021
+ }): Array<{
1022
+ candidate: T;
1023
+ score: number;
1024
+ }>;
1025
+ /**
1026
+ * NSGA-II crowding distance — secondary sort for ties on the frontier.
1027
+ *
1028
+ * When the Pareto front collapses to a single point (or many candidates tie
1029
+ * on dominance), naive selection picks arbitrarily and the population
1030
+ * degenerates over generations. NSGA-II preserves diversity by preferring
1031
+ * candidates with more empty space around them on the frontier.
1032
+ *
1033
+ * Returns an array of `{ candidate, distance }` in the SAME order as the
1034
+ * input. Higher distance = more isolated = should be preferred when
1035
+ * preserving diversity.
1036
+ */
1037
+ declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
1038
+ candidate: T;
1039
+ distance: number;
1040
+ }>;
1041
+ /**
1042
+ * Pareto frontier with tie-break by crowding distance — the canonical
1043
+ * NSGA-II selection step. Returns the frontier sorted by descending crowding
1044
+ * distance so callers can `.slice(0, k)` to pick K diverse winners.
1045
+ */
1046
+ declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
1047
+ candidate: T;
1048
+ distance: number;
1049
+ }>;
1050
+ //#endregion
1051
+ //#region src/campaign/gates/promotion-policy.d.ts
1052
+ /** Where an objective's per-cell scalar comes from. `composite` reads the
1053
+ * judge's composite; `dimension` reads a named per-dimension score. */
1054
+ type ObjectiveSource = {
1055
+ kind: 'composite';
1056
+ } | {
1057
+ kind: 'dimension';
1058
+ dimension: string;
1059
+ };
1060
+ interface PromotionObjective {
1061
+ /** Stable label used in reports + `contributingGates`. */
1062
+ name: string;
1063
+ source: ObjectiveSource;
1064
+ /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
1065
+ * the paired delta so a positive bootstrap always means "candidate better". */
1066
+ direction: Direction;
1067
+ /** The good-direction paired-delta CI lower bound must EXCEED this to count
1068
+ * as a significant gain on this axis. Interpreted in the judge's native
1069
+ * scale. Default 0 (⇒ "confidently better"). */
1070
+ gainThreshold?: number;
1071
+ /** A floor breach (regression) is declared when the good-direction CI lower
1072
+ * bound is below −floorTolerance. When omitted it auto-scales off observed
1073
+ * magnitudes (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
1074
+ floorTolerance?: number;
1075
+ }
1076
+ /** Per-axis verdict from the good-direction paired bootstrap. */
1077
+ type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
1078
+ interface AxisEvidence {
1079
+ name: string;
1080
+ source: ObjectiveSource;
1081
+ direction: Direction;
1082
+ /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
1083
+ * a positive value means the candidate is better on this axis. */
1084
+ bootstrap: PairedBootstrapResult;
1085
+ /** Paired observations contributing to this axis. */
1086
+ n: number;
1087
+ gainThreshold: number;
1088
+ floorTolerance: number;
1089
+ verdict: AxisVerdict;
1090
+ }
1091
+ interface EvidenceVector {
1092
+ /** One entry per objective — NOTHING averaged across axes. */
1093
+ axes: AxisEvidence[];
1094
+ /** Smallest paired n across axes that produced observations — the binding
1095
+ * evidence-sufficiency constraint. 0 when no axis produced observations. */
1096
+ minN: number;
1097
+ /** Aggregate per-side cost from the gate context (a constraint input, not a
1098
+ * CI axis — see the module header). */
1099
+ cost: {
1100
+ candidate: number;
1101
+ baseline: number;
1102
+ };
1103
+ }
1104
+ /** A promotion strategy: a pure function from the evidence vector to a verdict.
1105
+ * Many policies can run over the same `EvidenceVector` and disagree — that's
1106
+ * the point (competing strategies, shared evidence). */
1107
+ type PromotionPolicy = (ev: EvidenceVector) => GateResult;
1108
+ interface BuildEvidenceVectorOptions {
1109
+ /** Minimum paired observations before an axis can claim significance; below
1110
+ * it the axis is `few_runs`. Default 3. */
1111
+ minProductiveRuns?: number;
1112
+ /** Confidence level for every axis bootstrap. Default 0.95. */
1113
+ confidence?: number;
1114
+ /** Bootstrap resamples. Default 2000. */
1115
+ resamples?: number;
1116
+ /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
1117
+ seed?: number;
1118
+ }
1119
+ /**
1120
+ * The Evidence Bus. For each objective, pair candidate vs baseline by full
1121
+ * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
1122
+ * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
1123
+ * a single source of truth governs pairing granularity + scale handling.
1124
+ */
1125
+ declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
1126
+ /**
1127
+ * The default strategy: symmetric multi-objective Pareto significance. Ship iff
1128
+ * the candidate weakly dominates the baseline at the confidence level — no axis
1129
+ * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
1130
+ * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
1131
+ * need_more_work. Statistically equivalent → hold (never ship noise).
1132
+ */
1133
+ declare const paretoPolicy: PromotionPolicy;
1134
+ interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
1135
+ /** The objective vector. Every axis is both a gain source and a safety floor. */
1136
+ objectives: PromotionObjective[];
1137
+ /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
1138
+ * to run a stricter/looser strategy over the SAME bus (competing policies). */
1139
+ policy?: PromotionPolicy;
1140
+ /** Override the gate name in reports. */
1141
+ name?: string;
1142
+ }
1143
+ /**
1144
+ * Wrap the bus + a policy as a `Gate`. Plugs into the existing
1145
+ * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
1146
+ * loop behavior is unchanged because consumers opt in by passing this gate.
1147
+ */
1148
+ declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
1149
+ //#endregion
1150
+ //#region src/campaign/optimizer-model.d.ts
1151
+ type OptimizerModelBudget = ExternalOptimizerModelBudget;
1152
+ /** One metered OpenAI-compatible model connection shared by official optimizers. */
1153
+ interface OpenAICompatibleOptimizerModel {
1154
+ model: string;
1155
+ baseUrl: string;
1156
+ apiKey: string;
1157
+ budget: OptimizerModelBudget;
1158
+ }
1159
+ //#endregion
1160
+ //#region src/campaign/gepa-optimization-method.d.ts
1161
+ /** Shared settings for one bounded GEPA engine invocation. */
1162
+ interface GepaEngineOptions {
1163
+ /** GEPA engine name. GEPA validates names available in its Python runtime. */
1164
+ engine: string;
1165
+ /** Required cap for this engine's own model or CLI spend. */
1166
+ maxProposerCostUsd: number;
1167
+ /** Maximum concurrent evaluations inside this engine. Default: 1. */
1168
+ maxConcurrency?: number;
1169
+ /** Stop the engine after it reaches this score. */
1170
+ stopAtScore?: number;
1171
+ /** Isolate agent-based engines. Default: true. */
1172
+ sandbox?: boolean;
1173
+ /**
1174
+ * JSON-safe configuration for the registered GEPA engine.
1175
+ * Python callables and class instances cannot cross the process boundary.
1176
+ */
1177
+ engineConfig?: Record<string, unknown>;
1178
+ }
1179
+ /** One independently budgeted GEPA engine invocation. */
1180
+ interface GepaEngineRun extends GepaEngineOptions {
1181
+ /** Maximum callback evaluations this engine may consume. */
1182
+ maxEvaluations: number;
1183
+ }
1184
+ /** An engine in an adaptive run. All engines share the recipe evaluation limit. */
1185
+ type GepaAdaptiveEngineRun = GepaEngineOptions;
1186
+ /**
1187
+ * A direct mapping to a GEPA optimization recipe.
1188
+ *
1189
+ * GEPA owns every search and composition operation represented here. Tangle
1190
+ * supplies the candidate, data, execution callback, judges, and budgets.
1191
+ */
1192
+ type GepaOptimizationRecipe = {
1193
+ kind: 'engine';
1194
+ run: GepaEngineRun;
1195
+ } | {
1196
+ kind: 'sequential';
1197
+ runs: readonly GepaEngineRun[];
1198
+ } | {
1199
+ kind: 'adaptive-sequential';
1200
+ runs: readonly GepaAdaptiveEngineRun[];
1201
+ /** One evaluation budget shared by every adaptive stage. */
1202
+ maxEvaluations: number;
1203
+ /** Switch engines after this many evaluations without improvement. */
1204
+ plateauEvaluations: number;
1205
+ patience?: number;
1206
+ minEvaluationsPerStage?: number;
1207
+ improvementEpsilon?: number;
1208
+ cycle?: boolean;
1209
+ maxSwitches?: number;
1210
+ maxConcurrency?: number;
1211
+ } | {
1212
+ kind: 'best-of';
1213
+ runs: readonly GepaEngineRun[];
1214
+ maxWorkers?: number;
1215
+ } | {
1216
+ kind: 'vote';
1217
+ runs: readonly GepaEngineRun[];
1218
+ maxWorkers?: number;
1219
+ } | {
1220
+ kind: 'omni';
1221
+ explore: readonly GepaEngineRun[];
1222
+ continueWith: GepaEngineRun;
1223
+ maxWorkers?: number;
1224
+ };
1225
+ /** The command that runs the Python GEPA bridge. */
1226
+ type GepaRunnerCommand = ExternalOptimizerRunnerCommand;
1227
+ interface GepaOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
1228
+ /** Unique comparison-method name. Default identifies the GEPA recipe. */
1229
+ name?: string;
1230
+ /** A direct GEPA recipe. */
1231
+ recipe: GepaOptimizationRecipe;
1232
+ /** Plain-language goal shown to the external optimizer. */
1233
+ objective: string;
1234
+ /** Stable identity for the dispatch, judges, model settings, and scoring logic. */
1235
+ evaluationId: string;
1236
+ /** Optional bounded context about the surface and task. */
1237
+ background?: string;
1238
+ /**
1239
+ * Public dotted Python modules imported before GEPA resolves engine names.
1240
+ * Each module should call GEPA's official `register_engine()` API at import.
1241
+ */
1242
+ engineModules?: readonly string[];
1243
+ /** Reject external candidates longer than this. Default: 200,000 characters. */
1244
+ maxCandidateChars?: number;
1245
+ /** Reject serialized score evidence longer than this. Default: 100,000 characters. */
1246
+ maxEvidenceChars?: number;
1247
+ /** End the bridge process after this many milliseconds. Default: 30 minutes. */
1248
+ timeoutMs?: number;
1249
+ /**
1250
+ * OpenAI-compatible model used by standard GEPA reflection.
1251
+ * Calls pass through Agent Eval's local model proxy. Every recipe engine must
1252
+ * be `gepa` when this is set.
1253
+ */
1254
+ optimizer?: OpenAICompatibleOptimizerModel;
1255
+ /**
1256
+ * Decide what the external optimizer may read for a train or selection case.
1257
+ * The returned value must be JSON-serializable. The final comparison cases
1258
+ * are not accepted by this API and cannot be serialized here.
1259
+ */
1260
+ describeScenario?: (scenario: TScenario) => unknown;
1261
+ /** Optional bounded artifact evidence returned to GEPA after each evaluation. */
1262
+ describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
1263
+ /** Default: `never`. Compatible runs resume only when explicitly enabled. */
1264
+ resume?: ExternalOptimizerResumeMode;
1265
+ /**
1266
+ * Required for resumable direct GEPA runs because upstream state uses Python
1267
+ * pickle. Enable only for state created locally in a directory you control.
1268
+ */
1269
+ trustResumeState?: boolean;
1270
+ runner?: GepaRunnerCommand;
1271
+ }
1272
+ /**
1273
+ * Turn an optional GEPA installation into an `OptimizationMethod`.
1274
+ *
1275
+ * GEPA receives only serialized train and selection cases. The caller's final
1276
+ * test partition stays inside `compareOptimizationMethods`, which invokes this
1277
+ * method without a test-set field. The local callback routes every candidate
1278
+ * evaluation through the same dispatch and judges used by other methods.
1279
+ */
1280
+ declare function gepaOptimizationMethod<TScenario extends Scenario, TArtifact>(config: GepaOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
1281
+ //#endregion
1282
+ //#region src/campaign/presets/run-eval.d.ts
1283
+ interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
1284
+ runDir: string;
1285
+ }
1286
+ /**
1287
+ * Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
1288
+ */
1289
+ declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
1290
+ //#endregion
1291
+ //#region src/campaign/presets/run-optimization.d.ts
1292
+ interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
1293
+ /** Hash of the exact surface that produced `campaign`. */
1294
+ surfaceHash: string;
1295
+ /** Complete prior measurement reused by identity, including artifactsByPath. */
1296
+ campaign: CampaignResult<TArtifact, TScenario>;
1297
+ }
1298
+ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
1299
+ /** Initial mutable surface (typically system prompt or addendum). */
1300
+ baselineSurface: MutableSurface;
1301
+ /**
1302
+ * Complete prior measurement of `baselineSurface`. When present,
1303
+ * `runOptimization` validates its surface, scenario split, seed, reps, and
1304
+ * normal campaign coverage, then skips the baseline campaign entirely — no
1305
+ * dispatch or resumability-cache lookup. Candidate campaigns still run
1306
+ * normally. Prior spend remains in the imported campaign aggregates and is
1307
+ * not added again to this continuation's CostLedger.
1308
+ */
1309
+ premeasuredBaseline?: PremeasuredOptimizationBaseline<TArtifact, TScenario>;
1310
+ /** Dispatcher that takes the CURRENT surface + scenario → artifact. */
1311
+ dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
1312
+ /** The candidate-generation strategy. */
1313
+ proposer: SurfaceProposer;
1314
+ populationSize: number;
1315
+ maxGenerations: number;
1316
+ /** Candidate campaigns run at once. Default 1. Total concurrent cells are
1317
+ * bounded by candidateConcurrency * maxConcurrency. */
1318
+ candidateConcurrency?: number;
1319
+ /** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
1320
+ * agentic generator may take per candidate. */
1321
+ maxImprovementShots?: number;
1322
+ /** Optional analysis report forwarded to `propose()`. Opaque here; the
1323
+ * proposer types it. */
1324
+ report?: unknown;
1325
+ /** Structured findings forwarded to `propose()` as `ctx.findings`. A
1326
+ * findings producer emits these from the
1327
+ * generation's traces; findings-grounded proposers consume them. Opaque here;
1328
+ * the proposer types its `TFindings`. Empty when no producer is wired. */
1329
+ findings?: unknown[];
1330
+ /** Per-generation findings producer. Runs once on the BASELINE campaign
1331
+ * (as `generation: -1`, the baseline convention) before generation 0
1332
+ * proposes — so even a single-generation run proposes with trace context —
1333
+ * and then after each generation's candidates are scored with that
1334
+ * generation's results; whatever it returns REPLACES `ctx.findings` for the
1335
+ * NEXT `propose()`, so the diagnosis is refreshed each round instead
1336
+ * of being a static one-shot. Generic by design: the substrate does not
1337
+ * import an analyst — the consumer plugs its trace-analyst registry / HALO
1338
+ * here (reading the per-candidate `runDir` traces). When absent, findings
1339
+ * stay the static `opts.findings`. */
1340
+ analyzeGeneration?: (input: {
1341
+ generation: number;
1342
+ runDir: string;
1343
+ candidates: Array<{
1344
+ surfaceHash: string;
1345
+ campaign: CampaignResult<TArtifact, TScenario>;
1346
+ composite: number | null;
1347
+ }>;
1348
+ history: GenerationRecord[];
1349
+ /** Shared run spend account and receipt attribution phase. */
1350
+ costLedger?: CostLedgerHandle;
1351
+ costPhase?: string;
1352
+ }) => Promise<unknown[]>;
1353
+ /**
1354
+ * Optional override for how the WINNER is selected among coverage-complete
1355
+ * candidates (and how the incumbent bar is set). Returns a lexicographic rank
1356
+ * key — each element higher-is-better; candidates are ranked by descending key
1357
+ * (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
1358
+ * promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
1359
+ * scalar-mean ranking (single-element key ⇒ identical behavior).
1360
+ *
1361
+ * A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
1362
+ * instance resolved only when EVERY replicate resolved) passes a fail-closed
1363
+ * key built from the SAME reduction its gate uses, so winner-selection and the
1364
+ * ship-gate rank on the identical metric and can never invert — the selector
1365
+ * cannot promote a flaky per-cell-mean candidate the gate would reject over a
1366
+ * fail-closed candidate the gate would accept. Only the winner CHOICE changes;
1367
+ * the descriptive `composite` (mean) on every record and the Pareto objective
1368
+ * vectors are untouched, so proposer diversity and reporting are unaffected.
1369
+ */
1370
+ selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
1371
+ }
1372
+ type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
1373
+ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
1374
+ generations: Array<{
1375
+ record: GenerationRecord;
1376
+ surfaces: Array<{
1377
+ surfaceHash: string;
1378
+ surface: MutableSurface;
1379
+ campaign: CampaignResult<TArtifact, TScenario>;
1380
+ }>;
1381
+ }>;
1382
+ winnerSurface: MutableSurface;
1383
+ winnerSurfaceHash: string;
1384
+ /** Proposer label for the promoted surface. Present when the winning
1385
+ * candidate came from a `ProposedCandidate` (a reflective proposer);
1386
+ * absent when the winner is the baseline or a bare-surface mutator. */
1387
+ winnerLabel?: string;
1388
+ /** Proposer rationale for the promoted surface — the "because Z" that
1389
+ * motivated the winning change. Survives to `SelfImproveResult` and the
1390
+ * emitted provenance record. Absent when the winner is the baseline. */
1391
+ winnerRationale?: string;
1392
+ baselineCampaign: CampaignResult<TArtifact, TScenario>;
1393
+ /** Run-wide spend, including agents, proposers, analysts, and judges. */
1394
+ cost: CostLedgerSummary;
1395
+ /** The GEPA Pareto frontier across every scored surface (baseline + all
1396
+ * generations) by per-scenario objective vector — the non-dominated set.
1397
+ * Each generation's `propose()` received the frontier-so-far as
1398
+ * `ctx.paretoParents`; this is the final frontier. A surface here that is
1399
+ * NOT the winner is uniquely best on some scenario the winner loses on. */
1400
+ paretoFrontier: ParetoParent[];
1401
+ }
1402
+ /**
1403
+ * Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations.
1404
+ */
1405
+ declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
1406
+ //#endregion
1407
+ //#region src/campaign/presets/run-improvement-loop.d.ts
1408
+ type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
1409
+ /** Holdout scenarios kept OUT of the training optimization pool — used
1410
+ * ONLY to score baseline vs winner for the gate. */
1411
+ holdoutScenarios: TScenario[];
1412
+ /** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
1413
+ * `holdoutScenarios` and the gate decides on that held-out comparison.
1414
+ * `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
1415
+ * but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
1416
+ * the result + provenance record carry `holdout: 'deferred'` with NO
1417
+ * held-out lift — for callers that measure the held-out comparison in a
1418
+ * separate later run instead of faking a static holdout scenario and
1419
+ * recording a meaningless lift. */
1420
+ holdout?: 'measured' | 'deferred';
1421
+ /** Promotion gate. Substrate strongly recommends `defaultProductionGate`
1422
+ * for production wiring (composes red-team / reward-hacking / canary /
1423
+ * heldout). */
1424
+ gate: Gate<TArtifact, TScenario>;
1425
+ /** What to do when the gate ships:
1426
+ * - `'pr'`: open a PR via `openAutoPr`
1427
+ * - `'none'`: just report — caller decides what to do with the winner
1428
+ * Live-runtime self-mutation is intentionally unsupported. */
1429
+ autoOnPromote: 'pr' | 'none';
1430
+ /** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
1431
+ ghOwner?: string;
1432
+ ghRepo?: string;
1433
+ /** Placebo control. When supplied AND the winner differs from baseline, the
1434
+ * loop scores a THIRD holdout arm: the winner surface with its content
1435
+ * footprint-matched-blanked by this function (typically via `neutralizeText`).
1436
+ * Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
1437
+ * a `neutralizationGate` reject a win whose lift survives blanking the content
1438
+ * (decorative — driven by footprint, not content). Costs one extra holdout
1439
+ * campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
1440
+ neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
1441
+ };
1442
+ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
1443
+ baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
1444
+ winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
1445
+ neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
1446
+ neutralizedSurface?: MutableSurface;
1447
+ gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
1448
+ /** Present iff the loop ran with `holdout: 'deferred'`. When set,
1449
+ * `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
1450
+ * cells dispatched) and the gate verdict is the forced `'hold'`. */
1451
+ holdout?: 'deferred';
1452
+ /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
1453
+ * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
1454
+ * always present on the result + in the emitted provenance record. Empty
1455
+ * string when winner == baseline (no change to diff). */
1456
+ promotedDiff: string;
1457
+ prResult?: ReturnType<typeof openAutoPr>;
1458
+ }
1459
+ /**
1460
+ * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
1461
+ */
1462
+ declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
1463
+ //#endregion
1464
+ //#region src/campaign/provenance.d.ts
1465
+ interface LoopProvenanceCandidate {
1466
+ /** Generation index this candidate was proposed in. */
1467
+ generation: number;
1468
+ /** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
1469
+ surfaceHash: string;
1470
+ /** Full sha256 content hash — byte-identical-verifiable. */
1471
+ contentHash: string;
1472
+ /** Exact scored rows that produced this candidate's search result. */
1473
+ campaignDigest: `sha256:${string}`;
1474
+ /** Proposer label, when the proposer returned a `ProposedCandidate`. */
1475
+ label?: string;
1476
+ /** Proposer rationale — the "because Z". When the proposer returned a bare
1477
+ * surface (blind mutator) this is absent. */
1478
+ rationale?: string;
1479
+ /** Exact complete incumbent this candidate mutated. */
1480
+ parentSurfaceHash: string;
1481
+ /** Search-split composite of the exact parent. */
1482
+ parentComposite: number;
1483
+ /** Search-split composite change relative to the exact parent. */
1484
+ observedDeltaFromParent?: number;
1485
+ /** Whether the candidate completed every designed cell and could be selected. */
1486
+ eligibleForPromotion: boolean;
1487
+ /** Designed-denominator receipt retained even for incomplete candidates. */
1488
+ coverage: NonNullable<GenerationCandidate['coverage']>;
1489
+ /** Mean composite this candidate scored on the search split, or null when unscorable. */
1490
+ composite: number | null;
1491
+ /** Whether this candidate was promoted out of its generation. */
1492
+ promoted: boolean;
1493
+ }
1494
+ interface LoopProvenanceBackend {
1495
+ /** `assertRealBackend`-grade verdict over the worker call records. */
1496
+ verdict: 'real' | 'mixed' | 'stub';
1497
+ /** Number of worker LLM calls captured (the audit's "worker call count"). */
1498
+ workerCallCount: number;
1499
+ /** Distinct model ids observed across worker calls. */
1500
+ models: string[];
1501
+ totalInputTokens: number;
1502
+ totalOutputTokens: number;
1503
+ totalCostUsd: number;
1504
+ }
1505
+ interface LoopProvenanceEvidence {
1506
+ search: {
1507
+ splitDigest: `sha256:${string}`;
1508
+ baselineCampaignDigest: `sha256:${string}`;
1509
+ };
1510
+ holdout: {
1511
+ splitDigest: `sha256:${string}`;
1512
+ baselineCampaignDigest: `sha256:${string}`;
1513
+ winnerCampaignDigest: `sha256:${string}`;
1514
+ neutralized?: {
1515
+ contentHash: `sha256:${string}`;
1516
+ campaignDigest: `sha256:${string}`;
1517
+ composite: number;
1518
+ lift: number;
1519
+ };
1520
+ };
1521
+ costReceiptsDigest: `sha256:${string}`;
1522
+ }
1523
+ interface LoopProvenanceOptimizationMethod {
1524
+ name: string;
1525
+ cost: ComparisonCost;
1526
+ durationMs?: number;
1527
+ provenance?: OptimizationMethodProvenance;
1528
+ }
1529
+ /**
1530
+ * The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
1531
+ * ADDS the rationale + the explicit baseline→candidate diff (both omitted from
1532
+ * the bare hosted event) + backend provenance.
1533
+ */
1534
+ interface LoopProvenanceRecord {
1535
+ schema: 'tangle.loop-provenance';
1536
+ /** SHA-256 over the canonical record with this field omitted. */
1537
+ recordDigest: `sha256:${string}`;
1538
+ runId: string;
1539
+ runDir: string;
1540
+ timestamp: string;
1541
+ /** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
1542
+ baselineContentHash: string;
1543
+ winnerContentHash: string;
1544
+ /** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
1545
+ winnerLabel?: string;
1546
+ winnerRationale?: string;
1547
+ /** The explicit baseline→winner unified diff the gate decided on. */
1548
+ diff: string;
1549
+ /** Every candidate across every generation, with its rationale and structured cause. */
1550
+ candidates: LoopProvenanceCandidate[];
1551
+ /** Complete external method identity and spend, when one authored the candidate. */
1552
+ optimizationMethod?: LoopProvenanceOptimizationMethod;
1553
+ /** Exact campaign, split, surface, and receipt identities behind every summary. */
1554
+ evidence: LoopProvenanceEvidence;
1555
+ /** Baseline composite on the search split that generated the candidates. */
1556
+ baselineSearchComposite: number;
1557
+ /** The gate verdict — decision + reasons + contributing gates + delta. */
1558
+ gate: {
1559
+ decision: GateDecision;
1560
+ reasons: string[];
1561
+ delta?: number;
1562
+ contributingGates: GateContribution[];
1563
+ };
1564
+ /** Present iff the loop ran with `holdout: 'deferred'` — the held-out
1565
+ * comparison was intentionally not measured in this run, so the holdout
1566
+ * composites and `heldOutLift` are ABSENT rather than recorded as a
1567
+ * meaningless 0. */
1568
+ holdout?: 'deferred';
1569
+ /** baseline-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
1570
+ baselineHoldoutComposite?: number;
1571
+ /** winner-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
1572
+ winnerHoldoutComposite?: number;
1573
+ /** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. Absent
1574
+ * when `holdout === 'deferred'` (no held-out measurement ran). */
1575
+ heldOutLift?: number;
1576
+ /** Backend provenance: stub-vs-real verdict + worker call count + models. */
1577
+ backend: LoopProvenanceBackend;
1578
+ totalCostUsd: number;
1579
+ totalDurationMs: number;
1580
+ }
1581
+ interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
1582
+ runId: string;
1583
+ runDir: string;
1584
+ timestamp: string;
1585
+ baselineSurface: MutableSurface;
1586
+ winnerSurface: MutableSurface;
1587
+ winnerLabel?: string;
1588
+ winnerRationale?: string;
1589
+ /** Exact baseline campaign on the search split. */
1590
+ baselineSearchCampaign: CampaignResult<TArtifact, TScenario>;
1591
+ /** Per-generation candidate records straight off the loop result. */
1592
+ generations: Array<{
1593
+ generationIndex: number;
1594
+ candidates: GenerationCandidate[];
1595
+ promoted: string[];
1596
+ /** Surfaces measured this generation, keyed by surface hash so the content
1597
+ * hash can be computed and the loop identity rechecked from real bytes. */
1598
+ surfaces: Array<{
1599
+ surfaceHash: string;
1600
+ surface: MutableSurface;
1601
+ campaign: CampaignResult<TArtifact, TScenario>;
1602
+ }>;
1603
+ }>;
1604
+ gate: GateResult;
1605
+ /** Holdout policy the loop ran with. `'deferred'` ⇒ the holdout campaigns
1606
+ * below are the shared empty campaign and the record omits the holdout
1607
+ * composites + `heldOutLift`. Default `'measured'`. */
1608
+ holdout?: 'measured' | 'deferred';
1609
+ baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
1610
+ winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
1611
+ neutralizedSurface?: MutableSurface;
1612
+ neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
1613
+ /** Settled run-wide receipts — agent calls are the source for backend provenance. */
1614
+ costReceipts: ReadonlyArray<CostReceipt>;
1615
+ totalCostUsd: number;
1616
+ totalDurationMs: number;
1617
+ optimizationMethod?: LoopProvenanceOptimizationMethod;
1618
+ }
1619
+ interface LoopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario> {
1620
+ runId: string;
1621
+ runDir: string;
1622
+ timestamp: string;
1623
+ baselineSurface: MutableSurface;
1624
+ result: RunImprovementLoopResult<TArtifact, TScenario>;
1625
+ costReceipts: ReadonlyArray<CostReceipt>;
1626
+ totalCostUsd: number;
1627
+ totalDurationMs: number;
1628
+ }
1629
+ /** One translation from a completed improvement loop into durable evidence. */
1630
+ declare function loopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario>(input: LoopProvenanceArgsFromResult<TArtifact, TScenario>): BuildLoopProvenanceArgs<TArtifact, TScenario>;
1631
+ /** Build the durable provenance record from a completed loop result. */
1632
+ declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario>(args: BuildLoopProvenanceArgs<TArtifact, TScenario>): LoopProvenanceRecord;
1633
+ /** Digest the exact campaign fields that can affect a measured comparison. */
1634
+ declare function campaignMeasurementDigest<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): `sha256:${string}`;
1635
+ /** Recompute and validate the self-addressed durable record. */
1636
+ declare function verifyLoopProvenanceRecord(record: LoopProvenanceRecord): LoopProvenanceRecord;
1637
+ declare function canonicalDigest(value: unknown): `sha256:${string}`;
1638
+ /**
1639
+ * Build the loop's OTLP-ingestable spans from a provenance record. One root
1640
+ * span per loop (`tangle.runId`), one span per generation, one span per
1641
+ * candidate (carrying its surfaceHash + label), and one span for the gate
1642
+ * decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
1643
+ * the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
1644
+ * reads, so the hosted collector reconstructs the full tree.
1645
+ *
1646
+ * Times are synthesized monotonically off a single base so the span tree is
1647
+ * orderable; the substrate does not retain per-candidate wall-clock starts.
1648
+ */
1649
+ declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
1650
+ baseTimeMs?: number;
1651
+ }): TraceSpanEvent[];
1652
+ /** Canonical durable paths under the run dir. */
1653
+ declare function provenanceRecordPath(runDir: string): string;
1654
+ /**
1655
+ * Canonical path for the durable OTLP spans JSONL file under a loop run directory.
1656
+ */
1657
+ declare function provenanceSpansPath(runDir: string): string;
1658
+ interface EmitLoopProvenanceResult {
1659
+ record: LoopProvenanceRecord;
1660
+ spans: TraceSpanEvent[];
1661
+ /** Absolute paths the record + spans were written to, when storage persists. */
1662
+ recordPath: string;
1663
+ spansPath: string;
1664
+ }
1665
+ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends BuildLoopProvenanceArgs<TArtifact, TScenario> {
1666
+ /** Storage the record + spans are written through. */
1667
+ storage: CampaignStorage;
1668
+ /** When set, the spans are also shipped to the hosted `/v1/ingest/traces`
1669
+ * endpoint so the collector receives the full loop, not just `cost.*`. */
1670
+ hostedClient?: HostedClient;
1671
+ }
1672
+ /**
1673
+ * Build the provenance record + OTel spans and persist them durably under the
1674
+ * run dir (and ship spans to a hosted collector when one is wired). Returns
1675
+ * both artifacts so the caller can assert on / re-derive from them.
1676
+ *
1677
+ * Fail-loud: the durable write throws on storage failure (a swallowed write is
1678
+ * exactly the "emitted but lost" failure this closes). The hosted span ship is
1679
+ * the one best-effort leg — its failure is logged, not thrown, so an offline
1680
+ * collector never fails the loop (the durable artifact is the source of truth).
1681
+ */
1682
+ declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
1683
+ //#endregion
1684
+ //#region src/campaign/skillopt-optimization-method.d.ts
1685
+ interface SkillOptTrainerConfig {
1686
+ epochs: number;
1687
+ batchSize: number;
1688
+ accumulation?: number;
1689
+ editBudget?: number;
1690
+ minEditBudget?: number;
1691
+ analystWorkers?: number;
1692
+ minibatchSize?: number;
1693
+ mergeBatchSize?: number;
1694
+ maxAnalystRounds?: number;
1695
+ evaluationWorkers?: number;
1696
+ learningRateSchedule?: 'constant' | 'linear' | 'cosine' | 'autonomous';
1697
+ learningRateControl?: 'fixed' | 'autonomous' | 'none';
1698
+ updateMode?: 'patch' | 'rewrite_from_suggestions' | 'full_rewrite_minibatch';
1699
+ failureOnly?: boolean;
1700
+ useSlowUpdate?: boolean;
1701
+ useMetaSkill?: boolean;
1702
+ /**
1703
+ * Additional flat SkillOpt trainer settings. Tangle overwrites data,
1704
+ * output, split, seed, validation, and activation settings.
1705
+ */
1706
+ overrides?: Record<string, unknown>;
1707
+ }
1708
+ type SkillOptRunnerCommand = ExternalOptimizerRunnerCommand;
1709
+ interface SkillOptOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
1710
+ name?: string;
1711
+ /** Goal included with every described train and selection case. */
1712
+ objective: string;
1713
+ background?: string;
1714
+ /** Stable identity for the dispatch, judges, model settings, and scoring logic. */
1715
+ evaluationId: string;
1716
+ trainer: SkillOptTrainerConfig;
1717
+ /**
1718
+ * OpenAI-compatible model connection and hard limits for SkillOpt's own
1719
+ * optimizer calls.
1720
+ */
1721
+ optimizer: OpenAICompatibleOptimizerModel;
1722
+ /** Hard cap on candidate-case callback requests. */
1723
+ maxEvaluations: number;
1724
+ /** Scores at or above this value count as hard successes. Default: 1. */
1725
+ hardScoreThreshold?: number;
1726
+ maxCandidateChars?: number;
1727
+ /** Maximum serialized scenario plus evaluation evidence. Default: 100,000. */
1728
+ maxEvidenceChars?: number;
1729
+ timeoutMs?: number;
1730
+ describeScenario?: (scenario: TScenario) => unknown;
1731
+ describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
1732
+ resume?: ExternalOptimizerResumeMode;
1733
+ runner?: SkillOptRunnerCommand;
1734
+ }
1735
+ /** Run Microsoft's SkillOpt trainer as a complete optimization method. */
1736
+ declare function skillOptOptimizationMethod<TScenario extends Scenario, TArtifact>(config: SkillOptOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
1737
+ //#endregion
1738
+ export { Objective as $, optimizationTokenUsageFromSummary as $t, RunEvalOptions as A, CanarySeverity as At, OptimizerModelBudget as B, ComparisonCost as Bt, RunImprovementLoopOptions as C, ReferenceEquivalenceJudgeResult as Cn, scoreRedTeamOutput as Ct, RunOptimizationOptions as D, LlmJudgeDimension as Dn, CanaryKind as Dt, PremeasuredOptimizationBaseline as E, runReferenceEquivalenceJudge as En, CanaryEvaluation as Et, GepaOptimizationMethodConfig as F, ExternalTextOptimizerContext as Ft, ObjectiveSource as G, OptimizationMethodProvenance as Gt, AxisVerdict as H, OptimizationMethodComparison as Ht, GepaOptimizationRecipe as I, ExternalTextOptimizerResult as It, PromotionPolicy as J, OptimizationMethodScore as Jt, ParetoSignificanceGateOptions as K, OptimizationMethodResult as Kt, GepaRunnerCommand as L, ExternalOptimizationExample as Lt, GepaAdaptiveEngineRun as M, composeGate as Mt, GepaEngineOptions as N, externalTextOptimizationMethod as Nt, RunOptimizationResult as O, LlmJudgeOptions as On, CanaryOptions as Ot, GepaEngineRun as P, ExternalTextOptimizationMethodConfig as Pt, Direction as Q, costFromLedgerSummary as Qt, gepaOptimizationMethod as R, ExternalTextEvaluationResponse as Rt, verifyLoopProvenanceRecord as S, ReferenceEquivalenceJudgeOptions as Sn, redTeamReport as St, runImprovementLoop as T, createReferenceEquivalenceJudge as Tn, CanaryAlert as Tt, BuildEvidenceVectorOptions as U, OptimizationMethodInput as Ut, AxisEvidence as V, OptimizationMethod as Vt, EvidenceVector as W, OptimizationMethodPairwise as Wt, paretoPolicy as X, OptimizationTokenUsage as Xt, buildEvidenceVector as Y, OptimizationPackageSource as Yt, paretoSignificanceGate as Z, compareOptimizationMethods as Zt, emitLoopProvenance as _, OpenAutoPrResult as _n, RedTeamCategory as _t, BuildLoopProvenanceArgs as a, planCampaignRun as an, scalarScore as at, provenanceRecordPath as b, REFERENCE_EQUIVALENCE_JUDGE_VERSION as bn, RedTeamReport as bt, LoopProvenanceArgsFromResult as c, createRunCostLedger as cn, powerPreflight as ct, LoopProvenanceEvidence as d, assertCampaignDesign as dn, DefaultProductionGateCheck as dt, CampaignCellFailureReceipt as en, ParetoResult as et, LoopProvenanceOptimizationMethod as f, assertCampaignSplitIdentity as fn, DefaultProductionGateOptions as ft, canonicalDigest as g, OpenAutoPrOptions as gn, RedTeamCase as gt, campaignMeasurementDigest as h, campaignSplitDigestFromIdentities as hn, DEFAULT_RED_TEAM_CORPUS as ht, skillOptOptimizationMethod as i, RunCampaignOptions as in, paretoFrontierWithCrowding as it, runEval as j, runCanaries as jt, runOptimization as k, llmJudge as kn, CanaryReport as kt, LoopProvenanceBackend as l, fsCampaignStorage as ln, HeldOutGateOptions as lt, buildLoopProvenanceRecord as m, campaignSplitDigest as mn, defaultProductionGate as mt, SkillOptRunnerCommand as n, CampaignRunPlanCell as nn, dominates as nt, EmitLoopProvenanceArgs as o, runCampaign as on, PowerPreflight as ot, LoopProvenanceRecord as p, campaignScenarioIdentity as pn, DefaultProductionRewardHackingOptions as pt, PromotionObjective as q, OptimizationMethodRunOptions as qt, SkillOptTrainerConfig as r, PlanCampaignRunOptions as rn, paretoFrontier as rt, EmitLoopProvenanceResult as s, CampaignStorage as sn, PowerPreflightOptions as st, SkillOptOptimizationMethodConfig as t, CampaignRunPlan as tn, crowdingDistance as tt, LoopProvenanceCandidate as u, inMemoryCampaignStorage as un, heldOutGate as ut, loopProvenanceArgsFromResult as v, openAutoPr as vn, RedTeamFinding as vt, RunImprovementLoopResult as w, ReferenceEquivalenceScenario as wn, toolNamesForRun as wt, provenanceSpansPath as x, ReferenceEquivalenceJudgeInput as xn, redTeamDataset as xt, loopProvenanceSpans as y, REFERENCE_EQUIVALENCE_INPUT_LIMITS as yn, RedTeamPayload as yt, OpenAICompatibleOptimizerModel as z, CompareOptimizationMethodsOptions as zt };
1739
+ //# sourceMappingURL=skillopt-optimization-method-B7wX7XkF.d.ts.map