@tangle-network/agent-eval 0.163.2 → 0.170.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/CHANGELOG.md +114 -0
  2. package/README.md +2 -0
  3. package/dist/adapters/http.d.ts +108 -0
  4. package/dist/adapters/http.d.ts.map +1 -0
  5. package/dist/adapters/http.js +208 -0
  6. package/dist/adapters/http.js.map +1 -0
  7. package/dist/analyst/index.d.ts +40 -70
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +18 -311
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
  12. package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
  13. package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
  14. package/dist/benchmark-C4wk_Sjr.js.map +1 -0
  15. package/dist/{benchmark-command-CF-4GEWZ.js → benchmark-command-BA7qOdWw.js} +236 -251
  16. package/dist/benchmark-command-BA7qOdWw.js.map +1 -0
  17. package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
  18. package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
  19. package/dist/benchmarks/index.d.ts +5 -5
  20. package/dist/benchmarks/index.js +3 -3
  21. package/dist/builder-eval/index.d.ts +3 -3
  22. package/dist/builder-eval/index.js +1 -1
  23. package/dist/campaign/index.d.ts +8 -8
  24. package/dist/campaign/index.js +7 -7
  25. package/dist/{campaign-DQZmc2Dq.js → campaign-BeCbxFqs.js} +15 -14
  26. package/dist/campaign-BeCbxFqs.js.map +1 -0
  27. package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
  28. package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
  29. package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
  30. package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
  31. package/dist/cli.js +2 -2
  32. package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
  33. package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
  34. package/dist/{client-L9VVPkim.d.ts → client-_Fsa5c2_.d.ts} +4 -4
  35. package/dist/{client-L9VVPkim.d.ts.map → client-_Fsa5c2_.d.ts.map} +1 -1
  36. package/dist/contract/index.d.ts +12 -703
  37. package/dist/contract/index.js +11 -11
  38. package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
  39. package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
  40. package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-ovxrOP0_.d.ts} +6 -6
  41. package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-ovxrOP0_.d.ts.map} +1 -1
  42. package/dist/{define-agent-eval-D08pWIJb.js → define-agent-eval-Clj-8igZ.js} +20 -8
  43. package/dist/{define-agent-eval-D08pWIJb.js.map → define-agent-eval-Clj-8igZ.js.map} +1 -1
  44. package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-DVJm8Xlh.d.ts} +7 -7
  45. package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-DVJm8Xlh.d.ts.map} +1 -1
  46. package/dist/{dspy-rlm-engine-DhA9qKIm.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
  47. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
  48. package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
  49. package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
  50. package/dist/{engine-Cu5qD5Fc.d.ts → engine-D12Rb6WB.d.ts} +7 -7
  51. package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-D12Rb6WB.d.ts.map} +1 -1
  52. package/dist/{eval-campaign-BfohKmzx.js → eval-campaign-JDTeE6Pl.js} +4 -4
  53. package/dist/{eval-campaign-BfohKmzx.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
  54. package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
  55. package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
  56. package/dist/experiment/index.d.ts +5 -5
  57. package/dist/experiment/index.js +4 -4
  58. package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
  59. package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
  60. package/dist/{external-optimizer-process-BFmh36vW.js → external-optimizer-process-CQxylYeG.js} +4 -11
  61. package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
  62. package/dist/{external-optimizer-subprocess-CqLMW3nh.js → external-optimizer-subprocess-Cex8Da2i.js} +25 -11
  63. package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
  64. package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
  65. package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
  66. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
  67. package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
  68. package/dist/fuzz.d.ts +1 -1
  69. package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Bn7_xWCv.d.ts} +111 -111
  70. package/dist/heldout-gate-Bn7_xWCv.d.ts.map +1 -0
  71. package/dist/hosted/index.d.ts +2 -2
  72. package/dist/hosted/index.js +1 -1
  73. package/dist/index-Bfs5aufo.d.ts +704 -0
  74. package/dist/index-Bfs5aufo.d.ts.map +1 -0
  75. package/dist/{index-CGtH1piv.d.ts → index-CM-SM00y.d.ts} +7 -38
  76. package/dist/index-CM-SM00y.d.ts.map +1 -0
  77. package/dist/{index-D-V8gCs_.d.ts → index-DBbivBNs.d.ts} +30 -22
  78. package/dist/index-DBbivBNs.d.ts.map +1 -0
  79. package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
  80. package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
  81. package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
  82. package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
  83. package/dist/index.d.ts +67 -37
  84. package/dist/index.d.ts.map +1 -1
  85. package/dist/index.js +32 -26
  86. package/dist/index.js.map +1 -1
  87. package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
  88. package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
  89. package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
  90. package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
  91. package/dist/internal-BMFSR8Ns.js.map +1 -1
  92. package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
  93. package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
  94. package/dist/ledger-core/index.js +2 -2
  95. package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
  96. package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
  97. package/dist/{llm-judge-Du7WQPh7.js → llm-judge-DbJdo8Nj.js} +97 -19
  98. package/dist/llm-judge-DbJdo8Nj.js.map +1 -0
  99. package/dist/matrix/index.d.ts +2 -2
  100. package/dist/{matrix-eXKRMHnL.d.ts → matrix-BpI5Trmo.d.ts} +3 -3
  101. package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-BpI5Trmo.d.ts.map} +1 -1
  102. package/dist/meta-eval/index.d.ts +3 -3
  103. package/dist/meta-eval/index.js +1 -1
  104. package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
  105. package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
  106. package/dist/multishot/golden/index.d.ts +1 -1
  107. package/dist/multishot/index.d.ts +2 -2
  108. package/dist/openapi.json +4 -4
  109. package/dist/pipelines/index.d.ts +5 -5
  110. package/dist/pipelines/index.js +3 -3
  111. package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
  112. package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
  113. package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
  114. package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
  115. package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
  116. package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
  117. package/dist/{produced-state-Be0BK3RN.js → produced-state-CtSIp5cQ.js} +4 -4
  118. package/dist/{produced-state-Be0BK3RN.js.map → produced-state-CtSIp5cQ.js.map} +1 -1
  119. package/dist/profile-cell.js +1 -1
  120. package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-CkXSgKkF.d.ts} +3 -3
  121. package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-CkXSgKkF.d.ts.map} +1 -1
  122. package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
  123. package/dist/proposal-findings-bko3GGy-.js.map +1 -0
  124. package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CIRUardl.d.ts} +891 -891
  125. package/dist/provenance-CIRUardl.d.ts.map +1 -0
  126. package/dist/{query-_5g6re3_.js → query-BPGMVlbM.js} +3 -3
  127. package/dist/{query-_5g6re3_.js.map → query-BPGMVlbM.js.map} +1 -1
  128. package/dist/{query-CwnHlu5p.d.ts → query-Na5gEIGd.d.ts} +3 -3
  129. package/dist/{query-CwnHlu5p.d.ts.map → query-Na5gEIGd.d.ts.map} +1 -1
  130. package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
  131. package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
  132. package/dist/{release-confidence-nGDJiiwc.js → release-confidence-CzUHc4z4.js} +3 -3
  133. package/dist/{release-confidence-nGDJiiwc.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
  134. package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
  135. package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
  136. package/dist/reporting.d.ts +3 -3
  137. package/dist/reporting.js +2 -2
  138. package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
  139. package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
  140. package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
  141. package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
  142. package/dist/{reward-hacking-O5zKDANP.js → reward-hacking-SkxYgT0x.js} +2 -2
  143. package/dist/{reward-hacking-O5zKDANP.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
  144. package/dist/rl.d.ts +8 -8
  145. package/dist/rl.d.ts.map +1 -1
  146. package/dist/rl.js +6 -5
  147. package/dist/rl.js.map +1 -1
  148. package/dist/rollout/index.d.ts +1 -1
  149. package/dist/rollout/index.js +2 -2
  150. package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
  151. package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
  152. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
  153. package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
  154. package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
  155. package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
  156. package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
  157. package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
  158. package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
  159. package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
  160. package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
  161. package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
  162. package/dist/{semantic-concept-judge-BsDMOwJr.js → semantic-concept-judge-I36eejJx.js} +2 -2
  163. package/dist/{semantic-concept-judge-BsDMOwJr.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
  164. package/dist/{sequential-BLMbdrD7.js → sequential-B51qAYE4.js} +2 -2
  165. package/dist/{sequential-BLMbdrD7.js.map → sequential-B51qAYE4.js.map} +1 -1
  166. package/dist/{server-BjYiJHoJ.js → server-CCEnywOR.js} +26 -18
  167. package/dist/server-CCEnywOR.js.map +1 -0
  168. package/dist/{skillopt-optimization-method-UArRo-nr.js → skillopt-optimization-method-B2R9C5aG.js} +9 -9
  169. package/dist/{skillopt-optimization-method-UArRo-nr.js.map → skillopt-optimization-method-B2R9C5aG.js.map} +1 -1
  170. package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-DFS7QGpS.d.ts} +3 -3
  171. package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-DFS7QGpS.d.ts.map} +1 -1
  172. package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
  173. package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
  174. package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
  175. package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
  176. package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-B2DJ_82T.d.ts} +102 -36
  177. package/dist/store-tool-spans-B2DJ_82T.d.ts.map +1 -0
  178. package/dist/{store-tool-spans-BVga3c37.js → store-tool-spans-B9o6tU8f.js} +3 -3
  179. package/dist/{store-tool-spans-BVga3c37.js.map → store-tool-spans-B9o6tU8f.js.map} +1 -1
  180. package/dist/storyboard/index.d.ts +1 -1
  181. package/dist/{summary-report-BXeQ5Ues.js → summary-report-Bgh8CpNK.js} +2 -2
  182. package/dist/{summary-report-BXeQ5Ues.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
  183. package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
  184. package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
  185. package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
  186. package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
  187. package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-BnXlCJZQ.d.ts} +3 -3
  188. package/dist/tool-groups-BnXlCJZQ.d.ts.map +1 -0
  189. package/dist/{tool-waste-CKc7bYIg.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
  190. package/dist/{tool-waste-CKc7bYIg.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
  191. package/dist/{tool-waste-8BQiUc8K.js → tool-waste-CwGHzBzX.js} +2 -2
  192. package/dist/{tool-waste-8BQiUc8K.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
  193. package/dist/trace-repair/index.d.ts +3 -3
  194. package/dist/trace-repair/index.d.ts.map +1 -1
  195. package/dist/trace-repair/index.js +4 -3
  196. package/dist/trace-repair/index.js.map +1 -1
  197. package/dist/traces.d.ts +59 -61
  198. package/dist/traces.d.ts.map +1 -1
  199. package/dist/traces.js +7 -7
  200. package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
  201. package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
  202. package/dist/trajectory-replay/index.d.ts +3 -3
  203. package/dist/trajectory-replay/index.js +1 -1
  204. package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
  205. package/dist/types-CiWITkGo.js.map +1 -0
  206. package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
  207. package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
  208. package/dist/{types-D4s7Z6nq.d.ts → types-Dy237wiH.d.ts} +3 -3
  209. package/dist/{types-D4s7Z6nq.d.ts.map → types-Dy237wiH.d.ts.map} +1 -1
  210. package/dist/{types-BPb2Kf_C.d.ts → types-i21ccEkr.d.ts} +3 -3
  211. package/dist/{types-BPb2Kf_C.d.ts.map → types-i21ccEkr.d.ts.map} +1 -1
  212. package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
  213. package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
  214. package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
  215. package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
  216. package/dist/wire/index.d.ts +23 -8
  217. package/dist/wire/index.d.ts.map +1 -1
  218. package/dist/wire/index.js +2 -2
  219. package/docs/code-agent-intake.md +64 -0
  220. package/docs/concepts.md +1 -1
  221. package/docs/design/statistics-decisions.md +1 -1
  222. package/docs/distributed-driver.md +3 -6
  223. package/docs/public-api.md +122 -106
  224. package/docs/wire-protocol.md +5 -3
  225. package/package.json +9 -2
  226. package/dist/benchmark-BhT16ep9.js.map +0 -1
  227. package/dist/benchmark-command-CF-4GEWZ.js.map +0 -1
  228. package/dist/campaign-DQZmc2Dq.js.map +0 -1
  229. package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
  230. package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
  231. package/dist/contract/index.d.ts.map +0 -1
  232. package/dist/dspy-rlm-engine-DhA9qKIm.js.map +0 -1
  233. package/dist/external-optimizer-process-BFmh36vW.js.map +0 -1
  234. package/dist/external-optimizer-subprocess-CqLMW3nh.js.map +0 -1
  235. package/dist/index-CGtH1piv.d.ts.map +0 -1
  236. package/dist/index-D-V8gCs_.d.ts.map +0 -1
  237. package/dist/index-vrJugRal.d.ts +0 -1
  238. package/dist/kind-factory-DY8FdoXf.js.map +0 -1
  239. package/dist/llm-judge-Du7WQPh7.js.map +0 -1
  240. package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
  241. package/dist/run-score-lDzV0X8j.js.map +0 -1
  242. package/dist/server-BjYiJHoJ.js.map +0 -1
  243. package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
  244. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
  245. package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
  246. package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
  247. package/dist/types-BI4fT3HN.js.map +0 -1
@@ -1,13 +1,13 @@
1
1
  import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
2
2
  import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
3
- import { a as RunRecord } from "./run-record-VVy4T9OW.js";
3
+ import { a as RunRecord } from "./run-record-DQjRcYwA.js";
4
+ import { _ as ProposalFinding } from "./types-DMoNFDWi.js";
4
5
  import { p as ChatClient } from "./types-Bfk0uxRj.js";
5
- import { _ as ProposalFinding } from "./types-D9ssmxKL.js";
6
+ import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-Dy237wiH.js";
7
+ import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-lfaSeKSD.js";
6
8
  import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-szBJ_1vh.js";
7
- import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-D4s7Z6nq.js";
8
9
  import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
9
- import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-lfaSeKSD.js";
10
- import { g as TraceSpanEvent, t as HostedClient } from "./client-L9VVPkim.js";
10
+ import { g as TraceSpanEvent, t as HostedClient } from "./client-_Fsa5c2_.js";
11
11
  import { z } from "zod";
12
12
  //#region src/judge-families.d.ts
13
13
  /**
@@ -164,118 +164,6 @@ declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
164
164
  served: string | null | undefined;
165
165
  }>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
166
166
  //#endregion
167
- //#region src/llm-judge.d.ts
168
- /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
169
- * bare string uses the key as its own description. */
170
- type LlmJudgeDimension = string | JudgeDimension;
171
- interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
172
- /** The injected LLM transport. One `chat()` call per `score()`. Required —
173
- * there is no default route, so a misconfigured judge fails at construction,
174
- * never silently against the free-tier router. */
175
- chat: ChatClient;
176
- /** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
177
- * returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
178
- dimensions?: LlmJudgeDimension[];
179
- /** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
180
- model?: string;
181
- /** Explicit scoring revision for opaque transport or renderer changes. */
182
- judgeVersion?: string;
183
- temperature?: number;
184
- maxTokens?: number;
185
- /** Composite weights forwarded to `weightedComposite`: a partial map selects
186
- * AND weights exactly the named dimensions. Omit for a uniform mean. */
187
- weights?: Record<string, number>;
188
- /**
189
- * How to read a score out of the model's answer.
190
- *
191
- * `'sampled'` (default) reads the number the model emitted. Discrete grades
192
- * tie often, and a tie carries no ranking signal.
193
- *
194
- * `'expectation'` asks the provider for the log probabilities of the score
195
- * token and returns the expected value over the integer grades the model
196
- * considered, so two answers that both sample `8` separate by how much mass
197
- * sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
198
- * token, and a `unit` float is not. `whenUnavailable` decides what happens
199
- * when the provider returns no log probabilities, or the grade did not land
200
- * in one token: `'fail'` throws, `'sampled'` reads the emitted number and
201
- * records `scoringMethod: 'sampled'` on the score.
202
- */
203
- scoring?: {
204
- method: 'sampled';
205
- } | {
206
- method: 'expectation';
207
- whenUnavailable: 'fail' | 'sampled';
208
- };
209
- /** Scale the model is prompted to score on, normalized into `[0,1]`:
210
- * - `'unit'` (default): the model returns `[0,1]` directly.
211
- * - `'ten'`: the model returns `[0,10]`; divided by 10 here.
212
- * The prompt is annotated with the expected range either way. */
213
- scale?: 'unit' | 'ten';
214
- /** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
215
- appliesTo?: (scenario: TScenario) => boolean;
216
- /** Render the artifact + scenario into the user message. Default:
217
- * pretty-printed JSON of `{ scenario, artifact }`. */
218
- renderUser?: (input: {
219
- artifact: TArtifact;
220
- scenario: TScenario;
221
- }) => string;
222
- /** Strict runtime contract; its JSON Schema is sent to the provider. */
223
- costLedger?: CostLedgerHandle;
224
- responseSchema?: {
225
- name: string;
226
- schema: z.ZodObject;
227
- };
228
- }
229
- /**
230
- * Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
231
- * against `prompt` and reduces the model's per-dimension scores to a canonical
232
- * `JudgeScore` in `[0,1]`.
233
- *
234
- * The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
235
- * "notes": "…" }`; the helper strips fenced JSON, validates every declared
236
- * dimension is present and in range, normalizes by `scale`, and composites via
237
- * `weightedComposite`.
238
- */
239
- declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
240
- //#endregion
241
- //#region src/campaign/auto-pr.d.ts
242
- interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
243
- /** Campaign result to attach to the PR. */
244
- result: CampaignResult<TArtifact, TScenario>;
245
- /** Gate verdict explaining the promotion. Substrate refuses to open a PR
246
- * when `gate.decision !== 'ship'` — fails loud. */
247
- gate: GateResult;
248
- /** Promoted surface diff — typically the new system prompt addendum or
249
- * full profile diff. Substrate writes it as the PR body. */
250
- promotedDiff: string;
251
- /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
252
- ghOwner: string;
253
- ghRepo: string;
254
- /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
255
- branch?: string;
256
- /** PR title. Default includes manifest hash. */
257
- title?: string;
258
- /** Whether to actually open the PR or just dry-run. Default reads
259
- * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
260
- dryRun?: boolean;
261
- /** Test seam — substitute `gh pr create` invocation. */
262
- ghExec?: (args: string[]) => {
263
- stdout: string;
264
- stderr: string;
265
- status: number;
266
- };
267
- }
268
- interface OpenAutoPrResult {
269
- opened: boolean;
270
- prUrl?: string;
271
- dryRun: boolean;
272
- reason: string;
273
- }
274
- /**
275
- * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
276
- */
277
- declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
278
- //#endregion
279
167
  //#region src/campaign/storage.d.ts
280
168
  /**
281
169
  * `CampaignStorage` — the filesystem seam `runCampaign` writes through
@@ -326,6 +214,25 @@ declare function createRunCostLedger(input: {
326
214
  ensureRunDir?: boolean;
327
215
  }): CostLedger;
328
216
  //#endregion
217
+ //#region src/campaign/cell-schedule.d.ts
218
+ declare function cellCachePath(runDir: string, cellId: string): string;
219
+ //#endregion
220
+ //#region src/campaign/cell-cache.d.ts
221
+ type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
222
+ type CacheRead<TArtifact> = {
223
+ status: 'hit';
224
+ cell: CampaignCellResult<TArtifact>;
225
+ } | {
226
+ status: 'miss';
227
+ reason: CacheIssueReason;
228
+ };
229
+ declare function readCachedCell<TArtifact>(args: {
230
+ storage: CampaignStorage;
231
+ cachePath: string;
232
+ cellId: string;
233
+ manifestHash: string;
234
+ }): CacheRead<TArtifact>;
235
+ //#endregion
329
236
  //#region src/campaign/plan-campaign-run.d.ts
330
237
  interface CampaignRunPlanCell {
331
238
  cellId: string;
@@ -554,123 +461,86 @@ interface CampaignCellRetryPolicy {
554
461
  */
555
462
  declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
556
463
  //#endregion
557
- //#region src/campaign/cell-schedule.d.ts
558
- declare function cellCachePath(runDir: string, cellId: string): string;
559
- //#endregion
560
- //#region src/campaign/cell-cache.d.ts
561
- type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
562
- type CacheRead<TArtifact> = {
563
- status: 'hit';
564
- cell: CampaignCellResult<TArtifact>;
565
- } | {
566
- status: 'miss';
567
- reason: CacheIssueReason;
568
- };
569
- declare function readCachedCell<TArtifact>(args: {
570
- storage: CampaignStorage;
571
- cachePath: string;
572
- cellId: string;
573
- manifestHash: string;
574
- }): CacheRead<TArtifact>;
575
- //#endregion
576
- //#region src/campaign/external-optimizer-observations.d.ts
577
- interface ExternalOptimizerObservationSummary {
578
- scope: 'callback-submitted-candidates';
579
- path: string;
580
- sha256: `sha256:${string}`;
581
- submittedCandidates: number;
582
- evaluations: number;
583
- refusals: number;
584
- }
585
- interface ExternalOptimizerExecutionSummary {
586
- scope: 'runtime-model-calls';
587
- path: string;
588
- sha256: `sha256:${string}`;
589
- calls: number;
590
- succeeded: number;
591
- failed: number;
464
+ //#region src/campaign/presets/run-eval.d.ts
465
+ interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
466
+ runDir: string;
592
467
  }
593
- interface ExternalOptimizerSubmittedCandidate {
594
- /** Exact text or named-component surface submitted to the evaluation callback. */
595
- readonly candidate: ExternalTextCandidate;
596
- /** Eval's canonical content identity for `candidate`. */
597
- readonly candidateHash: string;
598
- readonly candidateDigest: `sha256:${string}`;
599
- readonly proposalSequence: number;
600
- /** Exact observation artifact that proves this candidate was submitted. */
601
- readonly provenance: {
602
- readonly path: string;
603
- readonly sha256: `sha256:${string}`;
468
+ /**
469
+ * Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
470
+ */
471
+ declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
472
+ //#endregion
473
+ //#region src/campaign/auto-pr.d.ts
474
+ interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
475
+ /** Campaign result to attach to the PR. */
476
+ result: CampaignResult<TArtifact, TScenario>;
477
+ /** Gate verdict explaining the promotion. Substrate refuses to open a PR
478
+ * when `gate.decision !== 'ship'` — fails loud. */
479
+ gate: GateResult;
480
+ /** Promoted surface diff — typically the new system prompt addendum or
481
+ * full profile diff. Substrate writes it as the PR body. */
482
+ promotedDiff: string;
483
+ /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
484
+ ghOwner: string;
485
+ ghRepo: string;
486
+ /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
487
+ branch?: string;
488
+ /** PR title. Default includes manifest hash. */
489
+ title?: string;
490
+ /** Whether to actually open the PR or just dry-run. Default reads
491
+ * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
492
+ dryRun?: boolean;
493
+ /** Test seam — substitute `gh pr create` invocation. */
494
+ ghExec?: (args: string[]) => {
495
+ stdout: string;
496
+ stderr: string;
497
+ status: number;
604
498
  };
605
499
  }
606
- interface ExternalOptimizerObservationArtifact {
607
- readonly summary: ExternalOptimizerObservationSummary;
608
- readonly observations: readonly ExternalOptimizerEvaluationObservation[];
609
- /** Every distinct callback-submitted candidate in proposal order. */
610
- readonly candidates: readonly ExternalOptimizerSubmittedCandidate[];
500
+ interface OpenAutoPrResult {
501
+ opened: boolean;
502
+ prUrl?: string;
503
+ dryRun: boolean;
504
+ reason: string;
611
505
  }
612
506
  /**
613
- * Read and verify the exact callback observation artifact addressed by method provenance.
614
- *
615
- * The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
616
- * identities, and summary counts before it returns any candidate.
617
- * This proves that the bytes match the supplied summary. The caller remains
618
- * responsible for obtaining that summary from trusted provenance.
507
+ * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
619
508
  */
620
- declare function readExternalOptimizerObservationArtifact(input: {
621
- summary: ExternalOptimizerObservationSummary;
622
- storage?: CampaignStorage;
623
- }): ExternalOptimizerObservationArtifact;
509
+ declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
624
510
  //#endregion
625
- //#region src/campaign/gepa-candidate-population.d.ts
626
- interface GepaCandidatePopulationSummary {
627
- readonly scope: 'gepa-candidate-population';
628
- readonly path: string;
629
- readonly sha256: `sha256:${string}`;
630
- readonly bytes: number;
631
- readonly runId: string;
632
- readonly candidates: number;
633
- readonly bestIndex: number;
634
- readonly maxCandidates: number;
635
- readonly maxCandidateChars: number;
636
- readonly scenarioIds: readonly string[];
637
- readonly surfaceKind: 'text' | 'components';
638
- }
639
- interface GepaCandidateSelectionScore {
640
- readonly scenarioId: string;
641
- readonly score: number;
642
- }
643
- interface GepaCandidatePopulationCandidate {
644
- /** Zero-based index assigned by the exact GEPA result. */
645
- readonly index: number;
646
- readonly candidate: ExternalTextCandidate;
647
- readonly candidateHash: string;
648
- readonly candidateDigest: `sha256:${string}`;
649
- /** Exact GEPA parent indices. The seed candidate has one null parent. */
650
- readonly parentIndices: readonly (number | null)[];
651
- /** Null means GEPA had no selection score for this candidate. */
652
- readonly aggregateScore: number | null;
653
- readonly selectionScores: readonly GepaCandidateSelectionScore[];
654
- readonly discoveryEvaluationCount: number;
511
+ //#region src/campaign/parent-selection.d.ts
512
+ /** Search state supplied to one parent-selection call. */
513
+ interface ParentSelectionContext {
514
+ /** Non-dominated scored surfaces across the whole run so far, including the
515
+ * baseline (`generation: -1`). Never empty. */
516
+ readonly frontier: ReadonlyArray<ParetoParent>;
517
+ /** Measured result of the global incumbent, the promotion bar. Under the
518
+ * default `selectionRankKey` the incumbent is always on `frontier`. */
519
+ readonly incumbent: ScoredSurfaceOutcome;
520
+ /** Every completed generation so far. */
521
+ readonly history: ReadonlyArray<GenerationRecord>;
522
+ /** Index of the generation about to propose. */
523
+ readonly generation: number;
655
524
  }
656
- interface GepaCandidatePopulationArtifact {
657
- readonly summary: GepaCandidatePopulationSummary;
658
- readonly runId: string;
659
- readonly bestIndex: number;
660
- readonly candidates: readonly GepaCandidatePopulationCandidate[];
525
+ /** Chooses the surface the next generation mutates. Returns one frontier
526
+ * parent; `runOptimization` refuses a parent it has not measured to
527
+ * completion or whose surface does not match its `surfaceHash`. */
528
+ type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
529
+ interface CrowdedFrontierParentOptions {
530
+ /** Integer seed for the per-generation draw. The same seed, frontier, and
531
+ * generation index select the same parent. */
532
+ seed: number;
661
533
  }
662
534
  /**
663
- * Read GEPA's exact candidate graph from the artifact addressed by method provenance.
664
- *
665
- * The reader checks the supplied digest, declared byte count, run identity,
666
- * candidate surfaces, parent graph, selection scores, and configured bounds.
667
- * This proves that the bytes match the supplied summary. The caller remains
668
- * responsible for obtaining that summary from trusted method provenance.
535
+ * NSGA-II crowded tournament selection over the frontier. Each generation
536
+ * draws two distinct frontier members with a PRNG seeded from `seed` and the
537
+ * generation index, and keeps the one with the larger crowding distance (more
538
+ * isolated on the frontier). Boundary parents carry infinite distance, so a
539
+ * boundary parent always beats an interior one. A tie on distance falls back
540
+ * to the higher mean composite, then to the smaller surface hash. A frontier
541
+ * of one member returns that member.
669
542
  */
670
- declare function readGepaCandidatePopulationArtifact(input: {
671
- summary: GepaCandidatePopulationSummary;
672
- storage?: CampaignStorage;
673
- }): GepaCandidatePopulationArtifact;
543
+ declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
674
544
  //#endregion
675
545
  //#region src/campaign/search-ledger.d.ts
676
546
  declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
@@ -1115,490 +985,184 @@ declare function assertCompleteSearchHistory(producerId: string, receipt: Search
1115
985
  /** Classify one producer's history without treating malformed evidence as absence. */
1116
986
  declare function searchHistoryCoverageRow(producerId: string, receipt: SearchHistoryReceipt | undefined): SearchHistoryCoverageRow;
1117
987
  //#endregion
1118
- //#region src/campaign/presets/compare-optimization-methods.d.ts
1119
- /** Shared campaign settings applied to every optimization method. */
1120
- type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
1121
- /** Cost reported by a method or by final test scoring. */
1122
- interface ComparisonCost {
1123
- /** Known subtotal. Consult `costProvenance` before treating this as total spend. */
1124
- totalCostUsd: number;
1125
- /** Exact origin of the total; uncaptured means `totalCostUsd` is only a known subtotal. */
1126
- costProvenance: CostProvenance;
1127
- accountingComplete: boolean;
1128
- incompleteReasons: string[];
988
+ //#region src/campaign/gepa-candidate-population.d.ts
989
+ interface GepaCandidatePopulationSummary {
990
+ readonly scope: 'gepa-candidate-population';
991
+ readonly path: string;
992
+ readonly sha256: `sha256:${string}`;
993
+ readonly bytes: number;
994
+ readonly runId: string;
995
+ readonly candidates: number;
996
+ readonly bestIndex: number;
997
+ readonly maxCandidates: number;
998
+ readonly maxCandidateChars: number;
999
+ readonly scenarioIds: readonly string[];
1000
+ readonly surfaceKind: 'text' | 'components';
1129
1001
  }
1130
- interface OptimizationPackageSource {
1131
- kind: 'package';
1132
- /** Whether package identity was inspected or supplied by caller code. */
1133
- evidence: 'observed' | 'declared';
1134
- package: string;
1135
- version: string;
1136
- sourceUrl?: string;
1137
- revision?: string;
1138
- /** SHA-256 of all installed module files observed before the run. */
1139
- sourceSha256?: string;
1002
+ interface GepaCandidateSelectionScore {
1003
+ readonly scenarioId: string;
1004
+ readonly score: number;
1140
1005
  }
1141
- interface OptimizationModuleSource {
1142
- module: string;
1143
- sourceSha256: string;
1006
+ interface GepaCandidatePopulationCandidate {
1007
+ /** Zero-based index assigned by the exact GEPA result. */
1008
+ readonly index: number;
1009
+ readonly candidate: ExternalTextCandidate;
1010
+ readonly candidateHash: string;
1011
+ readonly candidateDigest: `sha256:${string}`;
1012
+ /** Exact GEPA parent indices. The seed candidate has one null parent. */
1013
+ readonly parentIndices: readonly (number | null)[];
1014
+ /** Null means GEPA had no selection score for this candidate. */
1015
+ readonly aggregateScore: number | null;
1016
+ readonly selectionScores: readonly GepaCandidateSelectionScore[];
1017
+ readonly discoveryEvaluationCount: number;
1144
1018
  }
1145
- interface OptimizationPythonRuntime {
1146
- implementation: string;
1147
- version: string;
1019
+ interface GepaCandidatePopulationArtifact {
1020
+ readonly summary: GepaCandidatePopulationSummary;
1021
+ readonly runId: string;
1022
+ readonly bestIndex: number;
1023
+ readonly candidates: readonly GepaCandidatePopulationCandidate[];
1148
1024
  }
1149
- interface OptimizationTokenUsage {
1150
- /** All input tokens, including cache reads and cache creation. */
1151
- inputTokens: number;
1152
- /** Input tokens served from a provider cache. */
1153
- cachedInputTokens?: number;
1154
- /** Input tokens used to create or write a provider cache entry. */
1155
- cacheWriteInputTokens?: number;
1156
- outputTokens: number;
1157
- /** Reasoning tokens included in `outputTokens`. */
1158
- reasoningTokens?: number;
1159
- totalTokens: number;
1160
- calls: number;
1025
+ /**
1026
+ * Read GEPA's exact candidate graph from the artifact addressed by method provenance.
1027
+ *
1028
+ * The reader checks the supplied digest, declared byte count, run identity,
1029
+ * candidate surfaces, parent graph, selection scores, and configured bounds.
1030
+ * This proves that the bytes match the supplied summary. The caller remains
1031
+ * responsible for obtaining that summary from trusted method provenance.
1032
+ */
1033
+ declare function readGepaCandidatePopulationArtifact(input: {
1034
+ summary: GepaCandidatePopulationSummary;
1035
+ storage?: CampaignStorage;
1036
+ }): GepaCandidatePopulationArtifact;
1037
+ //#endregion
1038
+ //#region src/campaign/search-ledger-recording.d.ts
1039
+ /** How a search operation executed. The shape the ledger event records. */
1040
+ type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
1041
+ /** Immutable identities the ledger requires and a campaign cannot infer. */
1042
+ interface SearchRunIdentity {
1043
+ /** The agent implementation under optimization. */
1044
+ agent: SearchSourceRef;
1045
+ /** The candidate generator: a model call or deterministic code. */
1046
+ proposer: SearchExecutionIdentity;
1047
+ /** The code that plans the search and selects its winner. */
1048
+ search: SearchSourceRef;
1049
+ /** Model the agent runs. Used only for a cell that reported none. */
1050
+ model: SearchModelIdentity;
1161
1051
  }
1162
- interface OptimizationMethodProvenance {
1163
- /** External optimizer package. */
1164
- source: OptimizationPackageSource;
1165
- /** Python bridge package that invoked the optimizer. */
1166
- bridge?: OptimizationPackageSource;
1167
- /** Custom engine modules imported by the optimizer. */
1168
- modules?: OptimizationModuleSource[];
1169
- /** Python implementation used by the bridge process. */
1170
- python?: OptimizationPythonRuntime;
1171
- /** Exact model identifier configured for optimizer-owned model calls. */
1172
- optimizerModel?: string;
1173
- /** Stable public identity of the execution-owner callback. */
1174
- optimizerCallRef?: string;
1175
- runId: string;
1176
- /** Content identity shared by compatible resumptions. */
1177
- compatibleRunId?: string;
1178
- resumed: boolean;
1179
- /** Whether the run seed reached every external engine configuration. */
1180
- seedApplied?: boolean;
1181
- /** Evaluations the local callback metered — the trusted count. */
1182
- evaluationCount: number;
1183
- /**
1184
- * Evaluation total the external optimizer reported from its own counters.
1185
- * A difference from `evaluationCount` means upstream skipped, cached, or
1186
- * double-counted work; inspect before trusting upstream-derived budgets.
1187
- */
1188
- upstreamReportedEvaluations?: number;
1189
- artifactDir: string;
1190
- tokenUsage?: OptimizationTokenUsage;
1191
- /** Candidates submitted to the callback, per-case scores, and refusals. */
1192
- observations?: ExternalOptimizerObservationSummary;
1193
- /** Exact accepted GEPA candidates, parent indices, and selection scores. */
1194
- gepaCandidatePopulation?: GepaCandidatePopulationSummary;
1195
- /** Opaque Runtime execution evidence for every invoked optimizer-model call. */
1196
- modelExecutions?: ExternalOptimizerExecutionSummary;
1197
- /** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
1198
- anthropicEndpoint?: ExternalOptimizerWireCounts;
1052
+ interface SearchLedgerBinding {
1053
+ ledger: SearchLedger;
1054
+ identity: SearchRunIdentity;
1199
1055
  }
1200
- /** Shared inputs for one optimization method. Final test data is absent. */
1201
- interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
1202
- /** Surface every method starts from. */
1203
- readonly baselineSurface: MutableSurface;
1204
- /** Evidence used to author or fit candidates. */
1205
- readonly trainScenarios: readonly TScenario[];
1206
- /** Data used for candidate acceptance, early stopping, and model selection. */
1207
- readonly selectionScenarios: readonly TScenario[];
1208
- /** Runs one scenario with a candidate surface. */
1209
- readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1210
- /** Scores artifacts produced by `dispatchWithSurface`. */
1211
- readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
1212
- /** Method-specific artifacts are written below this directory. */
1213
- readonly runDir: string;
1214
- readonly seed: number;
1215
- /** Shared defaults for every method. A method may override them explicitly. */
1216
- readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
1217
- /** Durable spend account shared by every method and final scoring. */
1218
- readonly costLedger: CostLedgerHandle;
1056
+ /** One proposed candidate, before it is measured. */
1057
+ interface ProposedSearchCandidate {
1058
+ surface: MutableSurface;
1059
+ surfaceHash: string;
1060
+ label?: string;
1219
1061
  }
1220
- interface OptimizationMethodResult {
1221
- /** Surface selected without using the final test partition. */
1222
- winnerSurface: MutableSurface;
1223
- /** Optimization spend. Excludes final test scoring. */
1224
- cost: ComparisonCost;
1225
- /** Optimization duration. Excludes final test scoring. */
1226
- durationMs?: number;
1227
- /** Exact external implementation and run identity, when the method uses one. */
1228
- provenance?: OptimizationMethodProvenance;
1229
- /** Bounded proof envelope over the canonical SearchLedger for this optimization. */
1230
- searchHistory?: SearchHistoryReceipt;
1062
+ /** One measured candidate, after its campaign scored. */
1063
+ interface MeasuredSearchCandidate<TArtifact> {
1064
+ surface: MutableSurface;
1065
+ surfaceHash: string;
1066
+ cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
1067
+ runDir: string;
1068
+ /** False when the candidate missed a designed cell. */
1069
+ coverageComplete: boolean;
1231
1070
  }
1232
- /** A complete optimization method, including candidate generation and selection. */
1233
- interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
1234
- /** Unique, trimmed display name. Its normalized form must also be unique. */
1235
- name: string;
1236
- optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
1237
- }
1238
- interface OptimizationMethodScore {
1239
- name: string;
1240
- /** Mean final-test composite of the baseline (identical across methods). */
1241
- baselineComposite: number;
1242
- /** Mean final-test composite of this method's selected surface. */
1243
- winnerComposite: number;
1244
- /** Mean per-scenario final-test lift (winner minus baseline). */
1245
- lift: number;
1246
- /** Simultaneous paired-bootstrap interval for per-scenario lift.
1247
- * `low > 0` excludes zero after adjustment for all reported contrasts. */
1248
- liftCi: {
1249
- low: number;
1250
- high: number;
1251
- };
1252
- /** Optimization spend reported by the method. Excludes final test scoring. */
1253
- optimizationCost: ComparisonCost;
1254
- /** Optimization duration reported by the method. Excludes final test scoring. */
1255
- durationMs?: number;
1256
- /** Exact external implementation and run identity, when reported by the method. */
1257
- provenance?: OptimizationMethodProvenance;
1258
- /** Paired final-test values used to compute lift and its interval. */
1259
- scenarioScores: Array<{
1260
- scenarioId: string;
1261
- baselineComposite: number;
1262
- winnerComposite: number;
1263
- lift: number;
1264
- }>;
1265
- winnerSurface: MutableSurface;
1266
- /** 1-based, by descending lift. */
1267
- rank: number;
1268
- }
1269
- interface OptimizationMethodPairwise {
1270
- /** Higher-ranked method. */
1271
- a: string;
1272
- b: string;
1273
- /** Mean per-scenario untouched-test delta (a − b). */
1274
- deltaMean: number;
1275
- low: number;
1276
- high: number;
1277
- /** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
1278
- favored: string;
1279
- }
1280
- interface OptimizationMethodComparison {
1281
- /** Sorted by descending lift; `rank` set accordingly. */
1282
- scores: OptimizationMethodScore[];
1283
- best: OptimizationMethodScore;
1284
- /** Best vs each other method, using simultaneous paired-bootstrap intervals. */
1285
- pairwise: OptimizationMethodPairwise[];
1286
- testScenarioIds: string[];
1287
- /** Sum of the costs reported by every optimization method. */
1288
- optimizationCost: ComparisonCost;
1289
- /** Baseline and distinct winner scoring on the final test partition. */
1290
- testCost: ComparisonCost;
1291
- /** Optimization plus final test scoring. */
1292
- totalCost: ComparisonCost;
1293
- /** Caller-requested simultaneous coverage across all reported contrasts. */
1294
- confidence: number;
1295
- /** Bonferroni-adjusted confidence used for each bootstrap interval. */
1296
- intervalConfidence: number;
1297
- /** Method-vs-baseline plus all possible method-vs-method contrasts. */
1298
- comparisonCount: number;
1299
- /** Deterministic bootstrap and campaign seed. */
1300
- seed: number;
1301
- /** Bootstrap draws used for each interval. */
1302
- resamples: number;
1303
- /** Agent runs averaged within each test scenario before resampling scenarios. */
1304
- reps: number;
1305
- /** Coverage of every method's canonical search history. */
1306
- searchHistory: SearchHistoryCoverage;
1307
- }
1308
- interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
1309
- methods: OptimizationMethod<TScenario, TArtifact>[];
1310
- baselineSurface: MutableSurface;
1311
- /** Evidence used by every optimizer to author or fit candidates. */
1312
- trainScenarios: TScenario[];
1313
- /** Candidate acceptance, early-stopping, and optimizer-selection data. */
1314
- selectionScenarios: TScenario[];
1315
- /** Untouched final comparison data. Never passed to an optimization method. */
1316
- testScenarios: TScenario[];
1317
- /** Scores a surface on a scenario. The methods and final test share this function. */
1318
- dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1319
- judges: JudgeConfig<TArtifact, TScenario>[];
1320
- /** Bootstrap resamples for the lift intervals. Default is at least 2000 and
1321
- * rises when the requested simultaneous confidence needs finer tails. */
1322
- resamples?: number;
1323
- /** Shared defaults for each method's train and selection campaigns. */
1324
- optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
1325
- /** Number of optimization methods to run concurrently. Default 1. */
1326
- optimizationConcurrency?: number;
1327
- /** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
1328
- * Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
1329
- confidence?: number;
1330
- /** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
1331
- costCeiling?: number;
1332
- /**
1333
- * Missing history is reported by default. Publication-grade or autonomous
1334
- * callers set `require-complete`, which aborts before the first final-test call.
1335
- */
1336
- searchHistoryPolicy?: SearchHistoryPolicy;
1071
+ interface SearchRecorderOptions<TScenario extends Scenario> {
1072
+ binding: SearchLedgerBinding;
1073
+ storage: CampaignStorage;
1074
+ runDir: string;
1075
+ scenarios: ReadonlyArray<TScenario>;
1076
+ reps: number;
1077
+ maxGenerations: number;
1078
+ populationSize: number;
1079
+ /** Identity of the exact campaign design; the task benchmark pin. */
1080
+ splitDigest: `sha256:${string}`;
1081
+ /** Proposer label recorded on every candidate lineage. */
1082
+ proposerLabel: string;
1083
+ costLedger: CostLedgerHandle;
1337
1084
  }
1338
1085
  /**
1339
- * Compare complete optimization methods on disjoint train, selection, and final test data.
1086
+ * Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
1087
+ * and `recordResults()` per generation, then `finish()`.
1088
+ *
1089
+ * Every event id is derived from the run, and an id already durable is not
1090
+ * appended again, so a resumed run continues one ledger instead of conflicting
1091
+ * with its own history.
1340
1092
  */
1341
- declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
1342
- /** Keep the cost fields a custom optimization method must report. */
1343
- declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
1344
- /** Preserve every optimizer token class while keeping total input and output explicit. */
1345
- declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
1346
- /** Combine method costs without turning one unknown bill into a known total. */
1347
- declare function combineComparisonCosts(entries: ReadonlyArray<{
1348
- label: string;
1349
- cost: ComparisonCost;
1350
- }>): ComparisonCost;
1351
- //#endregion
1352
- //#region src/canary.d.ts
1353
- type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
1354
- type CanarySeverity = 'info' | 'warn' | 'error';
1355
- interface CanaryAlert {
1356
- kind: CanaryKind;
1357
- severity: CanarySeverity;
1358
- message: string;
1359
- /** Numbers that informed the decision — drop straight into a
1360
- * dashboard / paper figure. */
1361
- evidence: Record<string, unknown>;
1362
- }
1363
- interface CanaryReport {
1364
- alerts: CanaryAlert[];
1365
- /** Per-kind summary count. */
1366
- counts: Record<CanaryKind, number>;
1367
- /** Whether each enabled detector had enough observations to run. */
1368
- evaluations: CanaryEvaluation[];
1369
- }
1370
- interface CanaryEvaluation {
1371
- kind: CanaryKind;
1372
- status: 'evaluated' | 'not_evaluated';
1373
- observations: number;
1374
- reason?: string;
1375
- }
1376
- interface CanaryOptions {
1377
- /**
1378
- * Silent-fallback detection.
1379
- * - `constant`: confidence value treated as the fallback signal.
1380
- * Default 0.30 (matches the soft-fail default in
1381
- * `propose-review.ts`).
1382
- * - `consecutiveThreshold`: trip the alert after this many
1383
- * consecutive runs at `constant` (or `fallback === true`).
1384
- * Default 3.
1385
- */
1386
- silentFallback?: {
1387
- constant?: number;
1388
- consecutiveThreshold?: number;
1389
- /** Floating-point tolerance when comparing against `constant`. */
1390
- epsilon?: number;
1391
- };
1093
+ declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
1094
+ private readonly opts;
1095
+ private readonly tasks;
1096
+ private readonly registered;
1097
+ private readonly order;
1098
+ private readonly coverage;
1099
+ private readonly openSlots;
1100
+ private readonly durableEventIds;
1101
+ private lastStampMs;
1102
+ private proposalReceiptCount;
1103
+ private constructor();
1104
+ /** Open the recorder and append the plan. An existing ledger for the same
1105
+ * run is re-read first, so a resumed run keeps one plan and one lineage. */
1106
+ static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
1392
1107
  /**
1393
- * Calibration-drift detection.
1394
- * - `historyWindow`: number of past runs (oldest-first) treated as
1395
- * the historical baseline. Default 50.
1396
- * - `recentWindow`: number of recent runs (newest-first) compared
1397
- * against history. Default 20.
1398
- * - `ksAlpha`: alpha for the KS statistic vs critical value.
1399
- * Default 0.05.
1400
- * - `minRecent`: minimum recent runs required to even attempt the
1401
- * check. Default 10.
1108
+ * Record one generation's candidate-generation call and the candidates it
1109
+ * produced. A proposal larger than the planned population extends the plan
1110
+ * with the extra slots; a proposal that fills fewer closes the rest.
1402
1111
  */
1403
- calibrationDrift?: {
1404
- historyWindow?: number;
1405
- recentWindow?: number;
1406
- ksAlpha?: number;
1407
- minRecent?: number;
1408
- };
1112
+ recordGeneration(input: {
1113
+ generation: number;
1114
+ parentSurfaceHash: string;
1115
+ candidates: ReadonlyArray<ProposedSearchCandidate>;
1116
+ }): Promise<void>;
1117
+ /** Append one task attempt per designed cell of each candidate campaign. */
1118
+ recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
1409
1119
  /**
1410
- * Distribution-shift detection.
1411
- * - `category`: function that maps a run to a categorical bucket.
1412
- * Required to enable this canary; if omitted the chi-square check
1413
- * is skipped entirely.
1414
- * - `chiSquareAlpha`: alpha. Default 0.05.
1415
- * - `historyWindow`, `recentWindow`, `minRecent`: like above.
1120
+ * Close the search: unreached generations, the selection operation, one
1121
+ * decision per candidate, then the terminal event.
1122
+ *
1123
+ * The terminal event is appended only when canonical replay accounts for the
1124
+ * whole planned denominator. An interrupted or partly unscored search stays
1125
+ * `in-progress` and its receipt reports the exact gap, instead of claiming a
1126
+ * closed search.
1416
1127
  */
1417
- distributionShift?: {
1418
- category: (run: RunRecord) => string | null;
1419
- chiSquareAlpha?: number;
1420
- historyWindow?: number;
1421
- recentWindow?: number;
1422
- minRecent?: number;
1423
- };
1128
+ finish(input: {
1129
+ winnerSurfaceHash: string;
1130
+ generationsRun: number;
1131
+ runId: string;
1132
+ }): Promise<SearchHistoryReceipt>;
1133
+ /** Bounded receipt over the exact ledger bytes this run produced. */
1134
+ receipt(runId: string): Promise<SearchHistoryReceipt>;
1135
+ /** Read an existing ledger for this run so a resume continues it. */
1136
+ private hydrate;
1137
+ private plan;
1138
+ private registeredSlot;
1139
+ private remember;
1140
+ private recordGenerationOperation;
1141
+ private closeSlot;
1142
+ /** Spend booked to candidate generation since the previous generation. */
1143
+ private proposalAccounting;
1144
+ private cellModel;
1145
+ private proposalArtifact;
1146
+ /** Write one canonical evidence document and return its content address. */
1147
+ private writeArtifact;
1148
+ private append;
1149
+ /** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
1150
+ private stamp;
1424
1151
  }
1425
1152
  /**
1426
- * Run all configured canaries against a chronological run list.
1427
- * Runs MUST be sorted oldest-to-newest by the caller — the order of
1428
- * the input is used to define "recent" vs "historical" windows.
1429
- */
1430
- declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
1431
- //#endregion
1432
- //#region src/red-team.d.ts
1433
- type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
1434
- interface RedTeamPayload {
1435
- category: RedTeamCategory;
1436
- /** The hostile input the agent sees. */
1437
- input: string;
1438
- /**
1439
- * What a safe agent should do: "refuse" means explicit refusal,
1440
- * "ignore" means process benignly without executing the attack.
1441
- */
1442
- expected: 'refuse' | 'ignore';
1443
- /** Substring(s) that MUST NOT appear in output. */
1444
- forbiddenStrings?: string[];
1445
- /** Tool names the agent MUST NOT call for this scenario. */
1446
- forbiddenTools?: string[];
1447
- }
1448
- interface RedTeamCase extends DatasetScenario {
1449
- payload: RedTeamPayload;
1450
- }
1451
- interface RedTeamFinding {
1452
- scenarioId: string;
1453
- category: RedTeamCategory;
1454
- passed: boolean;
1455
- reason: string;
1456
- evidence?: string;
1457
- }
1458
- interface RedTeamReport {
1459
- findings: RedTeamFinding[];
1460
- passRateByCategory: Record<RedTeamCategory, number>;
1461
- overallPassRate: number;
1462
- }
1463
- /** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
1464
- declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
1465
- declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
1466
- /**
1467
- * Score a single agent response against the red-team case.
1468
- * `toolCalls` should be the names of tools the agent invoked during the run.
1469
- */
1470
- declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
1471
- /** Aggregate red-team findings into per-category pass rates. */
1472
- declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
1473
- //#endregion
1474
- //#region src/campaign/search-ledger-recording.d.ts
1475
- /** How a search operation executed. The shape the ledger event records. */
1476
- type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
1477
- /** Immutable identities the ledger requires and a campaign cannot infer. */
1478
- interface SearchRunIdentity {
1479
- /** The agent implementation under optimization. */
1480
- agent: SearchSourceRef;
1481
- /** The candidate generator: a model call or deterministic code. */
1482
- proposer: SearchExecutionIdentity;
1483
- /** The code that plans the search and selects its winner. */
1484
- search: SearchSourceRef;
1485
- /** Model the agent runs. Used only for a cell that reported none. */
1486
- model: SearchModelIdentity;
1487
- }
1488
- interface SearchLedgerBinding {
1489
- ledger: SearchLedger;
1490
- identity: SearchRunIdentity;
1491
- }
1492
- /** One proposed candidate, before it is measured. */
1493
- interface ProposedSearchCandidate {
1494
- surface: MutableSurface;
1495
- surfaceHash: string;
1496
- label?: string;
1497
- }
1498
- /** One measured candidate, after its campaign scored. */
1499
- interface MeasuredSearchCandidate<TArtifact> {
1500
- surface: MutableSurface;
1501
- surfaceHash: string;
1502
- cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
1503
- runDir: string;
1504
- /** False when the candidate missed a designed cell. */
1505
- coverageComplete: boolean;
1506
- }
1507
- interface SearchRecorderOptions<TScenario extends Scenario> {
1508
- binding: SearchLedgerBinding;
1509
- storage: CampaignStorage;
1510
- runDir: string;
1511
- scenarios: ReadonlyArray<TScenario>;
1512
- reps: number;
1513
- maxGenerations: number;
1514
- populationSize: number;
1515
- /** Identity of the exact campaign design; the task benchmark pin. */
1516
- splitDigest: `sha256:${string}`;
1517
- /** Proposer label recorded on every candidate lineage. */
1518
- proposerLabel: string;
1519
- costLedger: CostLedgerHandle;
1520
- }
1521
- /**
1522
- * Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
1523
- * and `recordResults()` per generation, then `finish()`.
1524
- *
1525
- * Every event id is derived from the run, and an id already durable is not
1526
- * appended again, so a resumed run continues one ledger instead of conflicting
1527
- * with its own history.
1528
- */
1529
- declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
1530
- private readonly opts;
1531
- private readonly tasks;
1532
- private readonly registered;
1533
- private readonly order;
1534
- private readonly coverage;
1535
- private readonly openSlots;
1536
- private readonly durableEventIds;
1537
- private lastStampMs;
1538
- private proposalReceiptCount;
1539
- private constructor();
1540
- /** Open the recorder and append the plan. An existing ledger for the same
1541
- * run is re-read first, so a resumed run keeps one plan and one lineage. */
1542
- static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
1543
- /**
1544
- * Record one generation's candidate-generation call and the candidates it
1545
- * produced. A proposal larger than the planned population extends the plan
1546
- * with the extra slots; a proposal that fills fewer closes the rest.
1547
- */
1548
- recordGeneration(input: {
1549
- generation: number;
1550
- parentSurfaceHash: string;
1551
- candidates: ReadonlyArray<ProposedSearchCandidate>;
1552
- }): Promise<void>;
1553
- /** Append one task attempt per designed cell of each candidate campaign. */
1554
- recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
1555
- /**
1556
- * Close the search: unreached generations, the selection operation, one
1557
- * decision per candidate, then the terminal event.
1558
- *
1559
- * The terminal event is appended only when canonical replay accounts for the
1560
- * whole planned denominator. An interrupted or partly unscored search stays
1561
- * `in-progress` and its receipt reports the exact gap, instead of claiming a
1562
- * closed search.
1563
- */
1564
- finish(input: {
1565
- winnerSurfaceHash: string;
1566
- generationsRun: number;
1567
- runId: string;
1568
- }): Promise<SearchHistoryReceipt>;
1569
- /** Bounded receipt over the exact ledger bytes this run produced. */
1570
- receipt(runId: string): Promise<SearchHistoryReceipt>;
1571
- /** Read an existing ledger for this run so a resume continues it. */
1572
- private hydrate;
1573
- private plan;
1574
- private registeredSlot;
1575
- private remember;
1576
- private recordGenerationOperation;
1577
- private closeSlot;
1578
- /** Spend booked to candidate generation since the previous generation. */
1579
- private proposalAccounting;
1580
- private cellModel;
1581
- private proposalArtifact;
1582
- /** Write one canonical evidence document and return its content address. */
1583
- private writeArtifact;
1584
- private append;
1585
- /** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
1586
- private stamp;
1587
- }
1588
- /**
1589
- * Record an optimizer's own candidate graph into the same ledger.
1590
- *
1591
- * A complete optimization method searches inside its own process and reports
1592
- * one artifact when it finishes: the candidate population, with each
1593
- * candidate's parents and its score per selection scenario. This turns that
1594
- * artifact into the canonical event stream, so a first-party method returns
1595
- * the same `SearchHistoryReceipt` the in-process loop returns, and
1596
- * `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
1597
- * accepts it.
1598
- *
1599
- * A candidate the optimizer left unscored on a planned scenario leaves the
1600
- * planned denominator open, so the receipt reports the gap instead of closing
1601
- * the search.
1153
+ * Record an optimizer's own candidate graph into the same ledger.
1154
+ *
1155
+ * A complete optimization method searches inside its own process and reports
1156
+ * one artifact when it finishes: the candidate population, with each
1157
+ * candidate's parents and its score per selection scenario. This turns that
1158
+ * artifact into the canonical event stream, so a first-party method returns
1159
+ * the same `SearchHistoryReceipt` the in-process loop returns, and
1160
+ * `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
1161
+ * accepts it.
1162
+ *
1163
+ * A candidate the optimizer left unscored on a planned scenario leaves the
1164
+ * planned denominator open, so the receipt reports the gap instead of closing
1165
+ * the search.
1602
1166
  */
1603
1167
  declare function recordCandidatePopulationSearch<TScenario extends Scenario>(input: {
1604
1168
  ledger: SearchLedger;
@@ -1614,49 +1178,6 @@ declare function recordCandidatePopulationSearch<TScenario extends Scenario>(inp
1614
1178
  runId: string;
1615
1179
  }): Promise<SearchHistoryReceipt>;
1616
1180
  //#endregion
1617
- //#region src/campaign/parent-selection.d.ts
1618
- /** Search state supplied to one parent-selection call. */
1619
- interface ParentSelectionContext {
1620
- /** Non-dominated scored surfaces across the whole run so far, including the
1621
- * baseline (`generation: -1`). Never empty. */
1622
- readonly frontier: ReadonlyArray<ParetoParent>;
1623
- /** Measured result of the global incumbent, the promotion bar. Under the
1624
- * default `selectionRankKey` the incumbent is always on `frontier`. */
1625
- readonly incumbent: ScoredSurfaceOutcome;
1626
- /** Every completed generation so far. */
1627
- readonly history: ReadonlyArray<GenerationRecord>;
1628
- /** Index of the generation about to propose. */
1629
- readonly generation: number;
1630
- }
1631
- /** Chooses the surface the next generation mutates. Returns one frontier
1632
- * parent; `runOptimization` refuses a parent it has not measured to
1633
- * completion or whose surface does not match its `surfaceHash`. */
1634
- type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
1635
- interface CrowdedFrontierParentOptions {
1636
- /** Integer seed for the per-generation draw. The same seed, frontier, and
1637
- * generation index select the same parent. */
1638
- seed: number;
1639
- }
1640
- /**
1641
- * NSGA-II crowded tournament selection over the frontier. Each generation
1642
- * draws two distinct frontier members with a PRNG seeded from `seed` and the
1643
- * generation index, and keeps the one with the larger crowding distance (more
1644
- * isolated on the frontier). Boundary parents carry infinite distance, so a
1645
- * boundary parent always beats an interior one. A tie on distance falls back
1646
- * to the higher mean composite, then to the smaller surface hash. A frontier
1647
- * of one member returns that member.
1648
- */
1649
- declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
1650
- //#endregion
1651
- //#region src/campaign/presets/run-eval.d.ts
1652
- interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
1653
- runDir: string;
1654
- }
1655
- /**
1656
- * Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
1657
- */
1658
- declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
1659
- //#endregion
1660
1181
  //#region src/campaign/presets/run-optimization.d.ts
1661
1182
  interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
1662
1183
  /** Hash of the exact surface that produced `campaign`. */
@@ -1714,146 +1235,653 @@ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> exte
1714
1235
  costPhase?: string;
1715
1236
  }) => Promise<ReadonlyArray<ProposalFinding>>;
1716
1237
  /**
1717
- * Optional override for how the WINNER is selected among coverage-complete
1718
- * candidates (and how the incumbent bar is set). Returns a lexicographic rank
1719
- * key — each element higher-is-better; candidates are ranked by descending key
1720
- * (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
1721
- * promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
1722
- * scalar-mean ranking (single-element key ⇒ identical behavior).
1723
- *
1724
- * A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
1725
- * instance resolved only when EVERY replicate resolved) passes a fail-closed
1726
- * key built from the SAME reduction its gate uses, so winner-selection and the
1727
- * ship-gate rank on the identical metric and can never invert — the selector
1728
- * cannot promote a flaky per-cell-mean candidate the gate would reject over a
1729
- * fail-closed candidate the gate would accept. Only the winner CHOICE changes;
1730
- * the descriptive `composite` (mean) on every record and the Pareto objective
1731
- * vectors are untouched, so proposer diversity and reporting are unaffected.
1238
+ * Optional override for how the WINNER is selected among coverage-complete
1239
+ * candidates (and how the incumbent bar is set). Returns a lexicographic rank
1240
+ * key — each element higher-is-better; candidates are ranked by descending key
1241
+ * (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
1242
+ * promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
1243
+ * scalar-mean ranking (single-element key ⇒ identical behavior).
1244
+ *
1245
+ * A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
1246
+ * instance resolved only when EVERY replicate resolved) passes a fail-closed
1247
+ * key built from the SAME reduction its gate uses, so winner-selection and the
1248
+ * ship-gate rank on the identical metric and can never invert — the selector
1249
+ * cannot promote a flaky per-cell-mean candidate the gate would reject over a
1250
+ * fail-closed candidate the gate would accept. Only the winner CHOICE changes;
1251
+ * the descriptive `composite` (mean) on every record and the Pareto objective
1252
+ * vectors are untouched, so proposer diversity and reporting are unaffected.
1253
+ */
1254
+ selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
1255
+ /**
1256
+ * Optional policy for which scored surface the next generation MUTATES.
1257
+ * Absent, every generation mutates the global incumbent, so the recorded
1258
+ * `parentSurfaceHash` lineage is a chain. Present, the selector receives the
1259
+ * Pareto frontier so far, the measured incumbent, the generation history,
1260
+ * and the generation index, and returns one frontier parent; the loop hands
1261
+ * that parent to `propose()` as `currentSurface` + `parentOutcome` and
1262
+ * records it as every candidate's `parentSurfaceHash`. Promotion is
1263
+ * unchanged: a candidate still has to beat the incumbent. The loop refuses
1264
+ * a parent it has not measured to completion. `crowdedFrontierParent` is
1265
+ * the provided seeded policy.
1266
+ */
1267
+ selectParent?: ParentSelector;
1268
+ /**
1269
+ * Record this search into a durable `SearchLedger`. The loop emits the plan,
1270
+ * each candidate-generation operation, each candidate registration with its
1271
+ * measured parent, one task attempt per designed cell, one decision per
1272
+ * candidate, and the terminal event, then returns a bounded
1273
+ * `searchHistory` receipt over the exact ledger bytes.
1274
+ *
1275
+ * `identity` declares what the ledger requires and a campaign cannot infer:
1276
+ * immutable revisions for the agent, proposer, and search implementations,
1277
+ * and the model the agent runs when a cell reports none.
1278
+ */
1279
+ searchLedger?: SearchLedgerBinding;
1280
+ }
1281
+ type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
1282
+ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
1283
+ generations: Array<{
1284
+ record: GenerationRecord;
1285
+ surfaces: Array<{
1286
+ surfaceHash: string;
1287
+ surface: MutableSurface;
1288
+ campaign: CampaignResult<TArtifact, TScenario>;
1289
+ }>;
1290
+ }>;
1291
+ /** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
1292
+ baselineSurface: MutableSurface;
1293
+ winnerSurface: MutableSurface;
1294
+ winnerSurfaceHash: string;
1295
+ /** Proposer label for the promoted surface. Present when the winning
1296
+ * candidate came from a `ProposedCandidate` (a reflective proposer);
1297
+ * absent when the winner is the baseline or a bare-surface mutator. */
1298
+ winnerLabel?: string;
1299
+ /** Proposer rationale for the promoted surface — the "because Z" that
1300
+ * motivated the winning change. Survives to `SelfImproveResult` and the
1301
+ * emitted provenance record. Absent when the winner is the baseline. */
1302
+ winnerRationale?: string;
1303
+ baselineCampaign: CampaignResult<TArtifact, TScenario>;
1304
+ /** Run-wide spend, including agents, proposers, analysts, and judges. */
1305
+ cost: CostLedgerSummary;
1306
+ /** Bounded proof envelope over the canonical search ledger. Present only
1307
+ * when `searchLedger` was supplied. `complete` is false when the search was
1308
+ * interrupted or a candidate left a designed cell unscored. */
1309
+ searchHistory?: SearchHistoryReceipt;
1310
+ /** The GEPA Pareto frontier across every scored surface (baseline + all
1311
+ * generations) by per-scenario objective vector — the non-dominated set.
1312
+ * Each generation's `propose()` received the frontier-so-far as
1313
+ * `ctx.paretoParents`; this is the final frontier. A surface here that is
1314
+ * NOT the winner is uniquely best on some scenario the winner loses on. */
1315
+ paretoFrontier: ParetoParent[];
1316
+ }
1317
+ /**
1318
+ * Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
1319
+ */
1320
+ declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
1321
+ //#endregion
1322
+ //#region src/campaign/presets/run-improvement-loop.d.ts
1323
+ type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
1324
+ /** Holdout scenarios kept OUT of the training optimization pool — used
1325
+ * ONLY to score baseline vs winner for the gate. */
1326
+ holdoutScenarios: TScenario[];
1327
+ /** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
1328
+ * `holdoutScenarios` and the gate decides on that held-out comparison.
1329
+ * `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
1330
+ * but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
1331
+ * the result + provenance record carry `holdout: 'deferred'` with NO
1332
+ * held-out lift — for callers that measure the held-out comparison in a
1333
+ * separate later run instead of faking a static holdout scenario and
1334
+ * recording a meaningless lift. */
1335
+ holdout?: 'measured' | 'deferred';
1336
+ /** Promotion gate. Substrate strongly recommends `defaultProductionGate`
1337
+ * for production wiring (composes red-team / reward-hacking / canary /
1338
+ * heldout). */
1339
+ gate: Gate<TArtifact, TScenario>;
1340
+ /** What to do when the gate ships:
1341
+ * - `'pr'`: open a PR via `openAutoPr`
1342
+ * - `'none'`: just report — caller decides what to do with the winner
1343
+ * Live-runtime self-mutation is intentionally unsupported. */
1344
+ autoOnPromote: 'pr' | 'none';
1345
+ /** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
1346
+ ghOwner?: string;
1347
+ ghRepo?: string;
1348
+ /** Placebo control. When supplied AND the winner differs from baseline, the
1349
+ * loop scores a THIRD holdout arm: the winner surface with its content
1350
+ * footprint-matched-blanked by this function (typically via `neutralizeText`).
1351
+ * Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
1352
+ * a `neutralizationGate` reject a win whose lift survives blanking the content
1353
+ * (decorative — driven by footprint, not content). Costs one extra holdout
1354
+ * campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
1355
+ neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
1356
+ };
1357
+ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
1358
+ baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
1359
+ winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
1360
+ neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
1361
+ neutralizedSurface?: MutableSurface;
1362
+ gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
1363
+ /** Present iff the loop ran with `holdout: 'deferred'`. When set,
1364
+ * `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
1365
+ * cells dispatched) and the gate verdict is the forced `'hold'`. */
1366
+ holdout?: 'deferred';
1367
+ /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
1368
+ * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
1369
+ * always present on the result + in the emitted provenance record. Empty
1370
+ * string when winner == baseline (no change to diff). */
1371
+ promotedDiff: string;
1372
+ prResult?: ReturnType<typeof openAutoPr>;
1373
+ }
1374
+ /**
1375
+ * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
1376
+ */
1377
+ declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
1378
+ //#endregion
1379
+ //#region src/campaign/transient-failure.d.ts
1380
+ interface TransientFailureOptions {
1381
+ /**
1382
+ * Treat full-duration timeouts ("timeout after 180000ms") as transient.
1383
+ * Enable on saturated shared infrastructure where queue starvation eats
1384
+ * the clock; leave off when the agent had the resources and simply failed.
1385
+ * Default false.
1386
+ */
1387
+ readonly retryFullDurationTimeouts?: boolean;
1388
+ /** Additional caller-specific transient patterns. */
1389
+ readonly extraPatterns?: readonly RegExp[];
1390
+ }
1391
+ /**
1392
+ * True when the error text describes an infrastructure hiccup that should be
1393
+ * retried rather than scored. Empty/undefined input is not transient.
1394
+ */
1395
+ declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
1396
+ /**
1397
+ * Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
1398
+ * failure whose error message `isTransientTransportFailure` classifies as an
1399
+ * infrastructure hiccup. A judge-stage failure is never retried here — the
1400
+ * dispatch already produced an artifact, so re-dispatching would score a
1401
+ * different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
1402
+ * is not transient by default; opt in via `extraPatterns` or
1403
+ * `retryFullDurationTimeouts` when queue starvation eats the clock.
1404
+ */
1405
+ declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
1406
+ //#endregion
1407
+ //#region src/llm-judge.d.ts
1408
+ /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
1409
+ * bare string uses the key as its own description. */
1410
+ type LlmJudgeDimension = string | JudgeDimension;
1411
+ interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
1412
+ /** The injected LLM transport. One `chat()` call per `score()`. Required —
1413
+ * there is no default route, so a misconfigured judge fails at construction,
1414
+ * never silently against the free-tier router. */
1415
+ chat: ChatClient;
1416
+ /** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
1417
+ * returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
1418
+ dimensions?: LlmJudgeDimension[];
1419
+ /** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
1420
+ model?: string;
1421
+ /** Explicit scoring revision for opaque transport or renderer changes. */
1422
+ judgeVersion?: string;
1423
+ temperature?: number;
1424
+ maxTokens?: number;
1425
+ /** Composite weights forwarded to `weightedComposite`: a partial map selects
1426
+ * AND weights exactly the named dimensions. Omit for a uniform mean. */
1427
+ weights?: Record<string, number>;
1428
+ /**
1429
+ * How to read a score out of the model's answer.
1430
+ *
1431
+ * `'sampled'` (default) reads the number the model emitted. Discrete grades
1432
+ * tie often, and a tie carries no ranking signal.
1433
+ *
1434
+ * `'expectation'` asks the provider for the log probabilities of the score
1435
+ * token and returns the expected value over the integer grades the model
1436
+ * considered, so two answers that both sample `8` separate by how much mass
1437
+ * sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
1438
+ * token, and a `unit` float is not. `whenUnavailable` decides what happens
1439
+ * when the provider returns no log probabilities, or the grade did not land
1440
+ * in one token: `'fail'` throws, `'sampled'` reads the emitted number and
1441
+ * records `scoringMethod: 'sampled'` on the score.
1442
+ */
1443
+ scoring?: {
1444
+ method: 'sampled';
1445
+ } | {
1446
+ method: 'expectation';
1447
+ whenUnavailable: 'fail' | 'sampled';
1448
+ };
1449
+ /** Scale the model is prompted to score on, normalized into `[0,1]`:
1450
+ * - `'unit'` (default): the model returns `[0,1]` directly.
1451
+ * - `'ten'`: the model returns `[0,10]`; divided by 10 here.
1452
+ * The prompt is annotated with the expected range either way. */
1453
+ scale?: 'unit' | 'ten';
1454
+ /** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
1455
+ appliesTo?: (scenario: TScenario) => boolean;
1456
+ /** Render the artifact + scenario into the user message. Default:
1457
+ * pretty-printed JSON of `{ scenario, artifact }`. */
1458
+ renderUser?: (input: {
1459
+ artifact: TArtifact;
1460
+ scenario: TScenario;
1461
+ }) => string;
1462
+ /** Strict runtime contract; its JSON Schema is sent to the provider. */
1463
+ costLedger?: CostLedgerHandle;
1464
+ responseSchema?: {
1465
+ name: string;
1466
+ schema: z.ZodObject;
1467
+ };
1468
+ }
1469
+ /**
1470
+ * Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
1471
+ * against `prompt` and reduces the model's per-dimension scores to a canonical
1472
+ * `JudgeScore` in `[0,1]`.
1473
+ *
1474
+ * The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
1475
+ * "notes": "…" }`; the helper strips fenced JSON, validates every declared
1476
+ * dimension is present and in range, normalizes by `scale`, and composites via
1477
+ * `weightedComposite`.
1478
+ */
1479
+ declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
1480
+ //#endregion
1481
+ //#region src/campaign/external-optimizer-observations.d.ts
1482
+ interface ExternalOptimizerObservationSummary {
1483
+ scope: 'callback-submitted-candidates';
1484
+ path: string;
1485
+ sha256: `sha256:${string}`;
1486
+ submittedCandidates: number;
1487
+ evaluations: number;
1488
+ refusals: number;
1489
+ }
1490
+ interface ExternalOptimizerExecutionSummary {
1491
+ scope: 'runtime-model-calls';
1492
+ path: string;
1493
+ sha256: `sha256:${string}`;
1494
+ calls: number;
1495
+ succeeded: number;
1496
+ failed: number;
1497
+ }
1498
+ interface ExternalOptimizerSubmittedCandidate {
1499
+ /** Exact text or named-component surface submitted to the evaluation callback. */
1500
+ readonly candidate: ExternalTextCandidate;
1501
+ /** Eval's canonical content identity for `candidate`. */
1502
+ readonly candidateHash: string;
1503
+ readonly candidateDigest: `sha256:${string}`;
1504
+ readonly proposalSequence: number;
1505
+ /** Exact observation artifact that proves this candidate was submitted. */
1506
+ readonly provenance: {
1507
+ readonly path: string;
1508
+ readonly sha256: `sha256:${string}`;
1509
+ };
1510
+ }
1511
+ interface ExternalOptimizerObservationArtifact {
1512
+ readonly summary: ExternalOptimizerObservationSummary;
1513
+ readonly observations: readonly ExternalOptimizerEvaluationObservation[];
1514
+ /** Every distinct callback-submitted candidate in proposal order. */
1515
+ readonly candidates: readonly ExternalOptimizerSubmittedCandidate[];
1516
+ }
1517
+ /**
1518
+ * Read and verify the exact callback observation artifact addressed by method provenance.
1519
+ *
1520
+ * The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
1521
+ * identities, and summary counts before it returns any candidate.
1522
+ * This proves that the bytes match the supplied summary. The caller remains
1523
+ * responsible for obtaining that summary from trusted provenance.
1524
+ */
1525
+ declare function readExternalOptimizerObservationArtifact(input: {
1526
+ summary: ExternalOptimizerObservationSummary;
1527
+ storage?: CampaignStorage;
1528
+ }): ExternalOptimizerObservationArtifact;
1529
+ //#endregion
1530
+ //#region src/campaign/presets/compare-optimization-methods.d.ts
1531
+ /** Shared campaign settings applied to every optimization method. */
1532
+ type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
1533
+ /** Cost reported by a method or by final test scoring. */
1534
+ interface ComparisonCost {
1535
+ /** Known subtotal. Consult `costProvenance` before treating this as total spend. */
1536
+ totalCostUsd: number;
1537
+ /** Exact origin of the total; uncaptured means `totalCostUsd` is only a known subtotal. */
1538
+ costProvenance: CostProvenance;
1539
+ accountingComplete: boolean;
1540
+ incompleteReasons: string[];
1541
+ }
1542
+ interface OptimizationPackageSource {
1543
+ kind: 'package';
1544
+ /** Whether package identity was inspected or supplied by caller code. */
1545
+ evidence: 'observed' | 'declared';
1546
+ package: string;
1547
+ version: string;
1548
+ sourceUrl?: string;
1549
+ revision?: string;
1550
+ /** SHA-256 of all installed module files observed before the run. */
1551
+ sourceSha256?: string;
1552
+ }
1553
+ interface OptimizationModuleSource {
1554
+ module: string;
1555
+ sourceSha256: string;
1556
+ }
1557
+ interface OptimizationPythonRuntime {
1558
+ implementation: string;
1559
+ version: string;
1560
+ }
1561
+ interface OptimizationTokenUsage {
1562
+ /** All input tokens, including cache reads and cache creation. */
1563
+ inputTokens: number;
1564
+ /** Input tokens served from a provider cache. */
1565
+ cachedInputTokens?: number;
1566
+ /** Input tokens used to create or write a provider cache entry. */
1567
+ cacheWriteInputTokens?: number;
1568
+ outputTokens: number;
1569
+ /** Reasoning tokens included in `outputTokens`. */
1570
+ reasoningTokens?: number;
1571
+ totalTokens: number;
1572
+ calls: number;
1573
+ }
1574
+ interface OptimizationMethodProvenance {
1575
+ /** External optimizer package. */
1576
+ source: OptimizationPackageSource;
1577
+ /** Python bridge package that invoked the optimizer. */
1578
+ bridge?: OptimizationPackageSource;
1579
+ /** Custom engine modules imported by the optimizer. */
1580
+ modules?: OptimizationModuleSource[];
1581
+ /** Python implementation used by the bridge process. */
1582
+ python?: OptimizationPythonRuntime;
1583
+ /** Exact model identifier configured for optimizer-owned model calls. */
1584
+ optimizerModel?: string;
1585
+ /** Stable public identity of the execution-owner callback. */
1586
+ optimizerCallRef?: string;
1587
+ runId: string;
1588
+ /** Content identity shared by compatible resumptions. */
1589
+ compatibleRunId?: string;
1590
+ resumed: boolean;
1591
+ /** Whether the run seed reached every external engine configuration. */
1592
+ seedApplied?: boolean;
1593
+ /** Evaluations the local callback metered — the trusted count. */
1594
+ evaluationCount: number;
1595
+ /**
1596
+ * Evaluation total the external optimizer reported from its own counters.
1597
+ * A difference from `evaluationCount` means upstream skipped, cached, or
1598
+ * double-counted work; inspect before trusting upstream-derived budgets.
1599
+ */
1600
+ upstreamReportedEvaluations?: number;
1601
+ artifactDir: string;
1602
+ tokenUsage?: OptimizationTokenUsage;
1603
+ /** Candidates submitted to the callback, per-case scores, and refusals. */
1604
+ observations?: ExternalOptimizerObservationSummary;
1605
+ /** Exact accepted GEPA candidates, parent indices, and selection scores. */
1606
+ gepaCandidatePopulation?: GepaCandidatePopulationSummary;
1607
+ /** Opaque Runtime execution evidence for every invoked optimizer-model call. */
1608
+ modelExecutions?: ExternalOptimizerExecutionSummary;
1609
+ /** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
1610
+ anthropicEndpoint?: ExternalOptimizerWireCounts;
1611
+ }
1612
+ /** Shared inputs for one optimization method. Final test data is absent. */
1613
+ interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
1614
+ /** Surface every method starts from. */
1615
+ readonly baselineSurface: MutableSurface;
1616
+ /** Evidence used to author or fit candidates. */
1617
+ readonly trainScenarios: readonly TScenario[];
1618
+ /** Data used for candidate acceptance, early stopping, and model selection. */
1619
+ readonly selectionScenarios: readonly TScenario[];
1620
+ /** Runs one scenario with a candidate surface. */
1621
+ readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1622
+ /** Scores artifacts produced by `dispatchWithSurface`. */
1623
+ readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
1624
+ /** Method-specific artifacts are written below this directory. */
1625
+ readonly runDir: string;
1626
+ readonly seed: number;
1627
+ /** Shared defaults for every method. A method may override them explicitly. */
1628
+ readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
1629
+ /** Durable spend account shared by every method and final scoring. */
1630
+ readonly costLedger: CostLedgerHandle;
1631
+ }
1632
+ interface OptimizationMethodResult {
1633
+ /** Surface selected without using the final test partition. */
1634
+ winnerSurface: MutableSurface;
1635
+ /** Optimization spend. Excludes final test scoring. */
1636
+ cost: ComparisonCost;
1637
+ /** Optimization duration. Excludes final test scoring. */
1638
+ durationMs?: number;
1639
+ /** Exact external implementation and run identity, when the method uses one. */
1640
+ provenance?: OptimizationMethodProvenance;
1641
+ /** Bounded proof envelope over the canonical SearchLedger for this optimization. */
1642
+ searchHistory?: SearchHistoryReceipt;
1643
+ }
1644
+ /** A complete optimization method, including candidate generation and selection. */
1645
+ interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
1646
+ /** Unique, trimmed display name. Its normalized form must also be unique. */
1647
+ name: string;
1648
+ optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
1649
+ }
1650
+ interface OptimizationMethodScore {
1651
+ name: string;
1652
+ /** Mean final-test composite of the baseline (identical across methods). */
1653
+ baselineComposite: number;
1654
+ /** Mean final-test composite of this method's selected surface. */
1655
+ winnerComposite: number;
1656
+ /** Mean per-scenario final-test lift (winner minus baseline). */
1657
+ lift: number;
1658
+ /** Simultaneous paired-bootstrap interval for per-scenario lift.
1659
+ * `low > 0` excludes zero after adjustment for all reported contrasts. */
1660
+ liftCi: {
1661
+ low: number;
1662
+ high: number;
1663
+ };
1664
+ /** Optimization spend reported by the method. Excludes final test scoring. */
1665
+ optimizationCost: ComparisonCost;
1666
+ /** Optimization duration reported by the method. Excludes final test scoring. */
1667
+ durationMs?: number;
1668
+ /** Exact external implementation and run identity, when reported by the method. */
1669
+ provenance?: OptimizationMethodProvenance;
1670
+ /** Paired final-test values used to compute lift and its interval. */
1671
+ scenarioScores: Array<{
1672
+ scenarioId: string;
1673
+ baselineComposite: number;
1674
+ winnerComposite: number;
1675
+ lift: number;
1676
+ }>;
1677
+ winnerSurface: MutableSurface;
1678
+ /** 1-based, by descending lift. */
1679
+ rank: number;
1680
+ }
1681
+ interface OptimizationMethodPairwise {
1682
+ /** Higher-ranked method. */
1683
+ a: string;
1684
+ b: string;
1685
+ /** Mean per-scenario untouched-test delta (a − b). */
1686
+ deltaMean: number;
1687
+ low: number;
1688
+ high: number;
1689
+ /** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
1690
+ favored: string;
1691
+ }
1692
+ interface OptimizationMethodComparison {
1693
+ /** Sorted by descending lift; `rank` set accordingly. */
1694
+ scores: OptimizationMethodScore[];
1695
+ best: OptimizationMethodScore;
1696
+ /** Best vs each other method, using simultaneous paired-bootstrap intervals. */
1697
+ pairwise: OptimizationMethodPairwise[];
1698
+ testScenarioIds: string[];
1699
+ /** Sum of the costs reported by every optimization method. */
1700
+ optimizationCost: ComparisonCost;
1701
+ /** Baseline and distinct winner scoring on the final test partition. */
1702
+ testCost: ComparisonCost;
1703
+ /** Optimization plus final test scoring. */
1704
+ totalCost: ComparisonCost;
1705
+ /** Caller-requested simultaneous coverage across all reported contrasts. */
1706
+ confidence: number;
1707
+ /** Bonferroni-adjusted confidence used for each bootstrap interval. */
1708
+ intervalConfidence: number;
1709
+ /** Method-vs-baseline plus all possible method-vs-method contrasts. */
1710
+ comparisonCount: number;
1711
+ /** Deterministic bootstrap and campaign seed. */
1712
+ seed: number;
1713
+ /** Bootstrap draws used for each interval. */
1714
+ resamples: number;
1715
+ /** Agent runs averaged within each test scenario before resampling scenarios. */
1716
+ reps: number;
1717
+ /** Coverage of every method's canonical search history. */
1718
+ searchHistory: SearchHistoryCoverage;
1719
+ }
1720
+ interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
1721
+ methods: OptimizationMethod<TScenario, TArtifact>[];
1722
+ baselineSurface: MutableSurface;
1723
+ /** Evidence used by every optimizer to author or fit candidates. */
1724
+ trainScenarios: TScenario[];
1725
+ /** Candidate acceptance, early-stopping, and optimizer-selection data. */
1726
+ selectionScenarios: TScenario[];
1727
+ /** Untouched final comparison data. Never passed to an optimization method. */
1728
+ testScenarios: TScenario[];
1729
+ /** Scores a surface on a scenario. The methods and final test share this function. */
1730
+ dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1731
+ judges: JudgeConfig<TArtifact, TScenario>[];
1732
+ /** Bootstrap resamples for the lift intervals. Default is at least 2000 and
1733
+ * rises when the requested simultaneous confidence needs finer tails. */
1734
+ resamples?: number;
1735
+ /** Shared defaults for each method's train and selection campaigns. */
1736
+ optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
1737
+ /** Number of optimization methods to run concurrently. Default 1. */
1738
+ optimizationConcurrency?: number;
1739
+ /** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
1740
+ * Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
1741
+ confidence?: number;
1742
+ /** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
1743
+ costCeiling?: number;
1744
+ /**
1745
+ * Missing history is reported by default. Publication-grade or autonomous
1746
+ * callers set `require-complete`, which aborts before the first final-test call.
1747
+ */
1748
+ searchHistoryPolicy?: SearchHistoryPolicy;
1749
+ }
1750
+ /**
1751
+ * Compare complete optimization methods on disjoint train, selection, and final test data.
1752
+ */
1753
+ declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
1754
+ /** Keep the cost fields a custom optimization method must report. */
1755
+ declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
1756
+ /** Preserve every optimizer token class while keeping total input and output explicit. */
1757
+ declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
1758
+ /** Combine method costs without turning one unknown bill into a known total. */
1759
+ declare function combineComparisonCosts(entries: ReadonlyArray<{
1760
+ label: string;
1761
+ cost: ComparisonCost;
1762
+ }>): ComparisonCost;
1763
+ //#endregion
1764
+ //#region src/canary.d.ts
1765
+ type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
1766
+ type CanarySeverity = 'info' | 'warn' | 'error';
1767
+ interface CanaryAlert {
1768
+ kind: CanaryKind;
1769
+ severity: CanarySeverity;
1770
+ message: string;
1771
+ /** Numbers that informed the decision — drop straight into a
1772
+ * dashboard / paper figure. */
1773
+ evidence: Record<string, unknown>;
1774
+ }
1775
+ interface CanaryReport {
1776
+ alerts: CanaryAlert[];
1777
+ /** Per-kind summary count. */
1778
+ counts: Record<CanaryKind, number>;
1779
+ /** Whether each enabled detector had enough observations to run. */
1780
+ evaluations: CanaryEvaluation[];
1781
+ }
1782
+ interface CanaryEvaluation {
1783
+ kind: CanaryKind;
1784
+ status: 'evaluated' | 'not_evaluated';
1785
+ observations: number;
1786
+ reason?: string;
1787
+ }
1788
+ interface CanaryOptions {
1789
+ /**
1790
+ * Silent-fallback detection.
1791
+ * - `constant`: confidence value treated as the fallback signal.
1792
+ * Default 0.30 (matches the soft-fail default in
1793
+ * `propose-review.ts`).
1794
+ * - `consecutiveThreshold`: trip the alert after this many
1795
+ * consecutive runs at `constant` (or `fallback === true`).
1796
+ * Default 3.
1732
1797
  */
1733
- selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
1798
+ silentFallback?: {
1799
+ constant?: number;
1800
+ consecutiveThreshold?: number;
1801
+ /** Floating-point tolerance when comparing against `constant`. */
1802
+ epsilon?: number;
1803
+ };
1734
1804
  /**
1735
- * Optional policy for which scored surface the next generation MUTATES.
1736
- * Absent, every generation mutates the global incumbent, so the recorded
1737
- * `parentSurfaceHash` lineage is a chain. Present, the selector receives the
1738
- * Pareto frontier so far, the measured incumbent, the generation history,
1739
- * and the generation index, and returns one frontier parent; the loop hands
1740
- * that parent to `propose()` as `currentSurface` + `parentOutcome` and
1741
- * records it as every candidate's `parentSurfaceHash`. Promotion is
1742
- * unchanged: a candidate still has to beat the incumbent. The loop refuses
1743
- * a parent it has not measured to completion. `crowdedFrontierParent` is
1744
- * the provided seeded policy.
1805
+ * Calibration-drift detection.
1806
+ * - `historyWindow`: number of past runs (oldest-first) treated as
1807
+ * the historical baseline. Default 50.
1808
+ * - `recentWindow`: number of recent runs (newest-first) compared
1809
+ * against history. Default 20.
1810
+ * - `ksAlpha`: alpha for the KS statistic vs critical value.
1811
+ * Default 0.05.
1812
+ * - `minRecent`: minimum recent runs required to even attempt the
1813
+ * check. Default 10.
1745
1814
  */
1746
- selectParent?: ParentSelector;
1815
+ calibrationDrift?: {
1816
+ historyWindow?: number;
1817
+ recentWindow?: number;
1818
+ ksAlpha?: number;
1819
+ minRecent?: number;
1820
+ };
1747
1821
  /**
1748
- * Record this search into a durable `SearchLedger`. The loop emits the plan,
1749
- * each candidate-generation operation, each candidate registration with its
1750
- * measured parent, one task attempt per designed cell, one decision per
1751
- * candidate, and the terminal event, then returns a bounded
1752
- * `searchHistory` receipt over the exact ledger bytes.
1753
- *
1754
- * `identity` declares what the ledger requires and a campaign cannot infer:
1755
- * immutable revisions for the agent, proposer, and search implementations,
1756
- * and the model the agent runs when a cell reports none.
1822
+ * Distribution-shift detection.
1823
+ * - `category`: function that maps a run to a categorical bucket.
1824
+ * Required to enable this canary; if omitted the chi-square check
1825
+ * is skipped entirely.
1826
+ * - `chiSquareAlpha`: alpha. Default 0.05.
1827
+ * - `historyWindow`, `recentWindow`, `minRecent`: like above.
1757
1828
  */
1758
- searchLedger?: SearchLedgerBinding;
1759
- }
1760
- type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
1761
- interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
1762
- generations: Array<{
1763
- record: GenerationRecord;
1764
- surfaces: Array<{
1765
- surfaceHash: string;
1766
- surface: MutableSurface;
1767
- campaign: CampaignResult<TArtifact, TScenario>;
1768
- }>;
1769
- }>;
1770
- /** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
1771
- baselineSurface: MutableSurface;
1772
- winnerSurface: MutableSurface;
1773
- winnerSurfaceHash: string;
1774
- /** Proposer label for the promoted surface. Present when the winning
1775
- * candidate came from a `ProposedCandidate` (a reflective proposer);
1776
- * absent when the winner is the baseline or a bare-surface mutator. */
1777
- winnerLabel?: string;
1778
- /** Proposer rationale for the promoted surface — the "because Z" that
1779
- * motivated the winning change. Survives to `SelfImproveResult` and the
1780
- * emitted provenance record. Absent when the winner is the baseline. */
1781
- winnerRationale?: string;
1782
- baselineCampaign: CampaignResult<TArtifact, TScenario>;
1783
- /** Run-wide spend, including agents, proposers, analysts, and judges. */
1784
- cost: CostLedgerSummary;
1785
- /** Bounded proof envelope over the canonical search ledger. Present only
1786
- * when `searchLedger` was supplied. `complete` is false when the search was
1787
- * interrupted or a candidate left a designed cell unscored. */
1788
- searchHistory?: SearchHistoryReceipt;
1789
- /** The GEPA Pareto frontier across every scored surface (baseline + all
1790
- * generations) by per-scenario objective vector — the non-dominated set.
1791
- * Each generation's `propose()` received the frontier-so-far as
1792
- * `ctx.paretoParents`; this is the final frontier. A surface here that is
1793
- * NOT the winner is uniquely best on some scenario the winner loses on. */
1794
- paretoFrontier: ParetoParent[];
1829
+ distributionShift?: {
1830
+ category: (run: RunRecord) => string | null;
1831
+ chiSquareAlpha?: number;
1832
+ historyWindow?: number;
1833
+ recentWindow?: number;
1834
+ minRecent?: number;
1835
+ };
1795
1836
  }
1796
1837
  /**
1797
- * Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
1838
+ * Run all configured canaries against a chronological run list.
1839
+ * Runs MUST be sorted oldest-to-newest by the caller — the order of
1840
+ * the input is used to define "recent" vs "historical" windows.
1798
1841
  */
1799
- declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
1842
+ declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
1800
1843
  //#endregion
1801
- //#region src/campaign/presets/run-improvement-loop.d.ts
1802
- type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
1803
- /** Holdout scenarios kept OUT of the training optimization pool — used
1804
- * ONLY to score baseline vs winner for the gate. */
1805
- holdoutScenarios: TScenario[];
1806
- /** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
1807
- * `holdoutScenarios` and the gate decides on that held-out comparison.
1808
- * `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
1809
- * but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
1810
- * the result + provenance record carry `holdout: 'deferred'` with NO
1811
- * held-out lift for callers that measure the held-out comparison in a
1812
- * separate later run instead of faking a static holdout scenario and
1813
- * recording a meaningless lift. */
1814
- holdout?: 'measured' | 'deferred';
1815
- /** Promotion gate. Substrate strongly recommends `defaultProductionGate`
1816
- * for production wiring (composes red-team / reward-hacking / canary /
1817
- * heldout). */
1818
- gate: Gate<TArtifact, TScenario>;
1819
- /** What to do when the gate ships:
1820
- * - `'pr'`: open a PR via `openAutoPr`
1821
- * - `'none'`: just report — caller decides what to do with the winner
1822
- * Live-runtime self-mutation is intentionally unsupported. */
1823
- autoOnPromote: 'pr' | 'none';
1824
- /** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
1825
- ghOwner?: string;
1826
- ghRepo?: string;
1827
- /** Placebo control. When supplied AND the winner differs from baseline, the
1828
- * loop scores a THIRD holdout arm: the winner surface with its content
1829
- * footprint-matched-blanked by this function (typically via `neutralizeText`).
1830
- * Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
1831
- * a `neutralizationGate` reject a win whose lift survives blanking the content
1832
- * (decorative — driven by footprint, not content). Costs one extra holdout
1833
- * campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
1834
- neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
1835
- };
1836
- interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
1837
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
1838
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
1839
- neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
1840
- neutralizedSurface?: MutableSurface;
1841
- gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
1842
- /** Present iff the loop ran with `holdout: 'deferred'`. When set,
1843
- * `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
1844
- * cells dispatched) and the gate verdict is the forced `'hold'`. */
1845
- holdout?: 'deferred';
1846
- /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
1847
- * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
1848
- * always present on the result + in the emitted provenance record. Empty
1849
- * string when winner == baseline (no change to diff). */
1850
- promotedDiff: string;
1851
- prResult?: ReturnType<typeof openAutoPr>;
1844
+ //#region src/red-team.d.ts
1845
+ type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
1846
+ interface RedTeamPayload {
1847
+ category: RedTeamCategory;
1848
+ /** The hostile input the agent sees. */
1849
+ input: string;
1850
+ /**
1851
+ * What a safe agent should do: "refuse" means explicit refusal,
1852
+ * "ignore" means process benignly without executing the attack.
1853
+ */
1854
+ expected: 'refuse' | 'ignore';
1855
+ /** Substring(s) that MUST NOT appear in output. */
1856
+ forbiddenStrings?: string[];
1857
+ /** Tool names the agent MUST NOT call for this scenario. */
1858
+ forbiddenTools?: string[];
1859
+ }
1860
+ interface RedTeamCase extends DatasetScenario {
1861
+ payload: RedTeamPayload;
1862
+ }
1863
+ interface RedTeamFinding {
1864
+ scenarioId: string;
1865
+ category: RedTeamCategory;
1866
+ passed: boolean;
1867
+ reason: string;
1868
+ evidence?: string;
1869
+ }
1870
+ interface RedTeamReport {
1871
+ findings: RedTeamFinding[];
1872
+ passRateByCategory: Record<RedTeamCategory, number>;
1873
+ overallPassRate: number;
1852
1874
  }
1875
+ /** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
1876
+ declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
1877
+ declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
1853
1878
  /**
1854
- * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
1879
+ * Score a single agent response against the red-team case.
1880
+ * `toolCalls` should be the names of tools the agent invoked during the run.
1855
1881
  */
1856
- declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
1882
+ declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
1883
+ /** Aggregate red-team findings into per-category pass rates. */
1884
+ declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
1857
1885
  //#endregion
1858
1886
  //#region src/campaign/provenance.d.ts
1859
1887
  interface LoopProvenanceCandidate {
@@ -2077,33 +2105,5 @@ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends
2077
2105
  */
2078
2106
  declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
2079
2107
  //#endregion
2080
- //#region src/campaign/transient-failure.d.ts
2081
- interface TransientFailureOptions {
2082
- /**
2083
- * Treat full-duration timeouts ("timeout after 180000ms") as transient.
2084
- * Enable on saturated shared infrastructure where queue starvation eats
2085
- * the clock; leave off when the agent had the resources and simply failed.
2086
- * Default false.
2087
- */
2088
- readonly retryFullDurationTimeouts?: boolean;
2089
- /** Additional caller-specific transient patterns. */
2090
- readonly extraPatterns?: readonly RegExp[];
2091
- }
2092
- /**
2093
- * True when the error text describes an infrastructure hiccup that should be
2094
- * retried rather than scored. Empty/undefined input is not transient.
2095
- */
2096
- declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
2097
- /**
2098
- * Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
2099
- * failure whose error message `isTransientTransportFailure` classifies as an
2100
- * infrastructure hiccup. A judge-stage failure is never retried here — the
2101
- * dispatch already produced an artifact, so re-dispatching would score a
2102
- * different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
2103
- * is not transient by default; opt in via `extraPatterns` or
2104
- * `retryFullDurationTimeouts` when queue starvation eats the clock.
2105
- */
2106
- declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
2107
- //#endregion
2108
- export { CanaryKind as $, ServedModelVerdict as $n, SearchOperationKind as $t, runEval as A, CampaignCellRetryPolicy as An, verifySearchHistoryReceipt as At, SearchRecorderOptions as B, inMemoryCampaignStorage as Bn, SearchCandidateSlotClosedEvent as Bt, RunImprovementLoopResult as C, ExternalOptimizerSubmittedCandidate as Cn, SearchHistoryPolicy as Ct, RunOptimizationResult as D, readCachedCell as Dn, assertSearchHistoryMatchesReplay as Dt, RunOptimizationOptions as E, CacheRead as En, assertCompleteSearchHistory as Et, MeasuredSearchCandidate as F, PlanCampaignRunOptions as Fn, SearchAttemptAccounting as Ft, RedTeamCategory as G, LlmJudgeOptions as Gn, SearchLedger as Gt, recordCandidatePopulationSearch as H, OpenAutoPrResult as Hn, SearchCompletedEvent as Ht, ProposedSearchCandidate as I, planCampaignRun as In, SearchCandidateDecidedEvent as It, redTeamDataset as J, AssertServedModelOptions as Jn, SearchLedgerEvent as Jt, RedTeamFinding as K, llmJudge as Kn, SearchLedgerAppendResult as Kt, SearchExecutionIdentity as L, CampaignStorage as Ln, SearchCandidateLineage as Lt, ParentSelectionContext as M, runCampaign as Mn, OpenSearchLedgerOptions as Mt, ParentSelector as N, CampaignRunPlan as Nn, SearchAccountingAudit as Nt, runOptimization as O, cellCachePath as On, createSearchHistoryReceipt as Ot, crowdedFrontierParent as P, CampaignRunPlanCell as Pn, SearchArtifactRef as Pt, CanaryEvaluation as Q, ServedModelPolicy as Qn, SearchModelIdentity as Qt, SearchLedgerBinding as R, createRunCostLedger as Rn, SearchCandidateRegisteredEvent as Rt, RunImprovementLoopOptions as S, ExternalOptimizerObservationSummary as Sn, SearchHistoryCoverageRow as St, PremeasuredOptimizationBaseline as T, CacheIssueReason as Tn, SearchHistoryRequiredError as Tt, DEFAULT_RED_TEAM_CORPUS as U, openAutoPr as Un, SearchCostAccounting as Ut, SearchRunIdentity as V, OpenAutoPrOptions as Vn, SearchCandidateSurface as Vt, RedTeamCase as W, LlmJudgeDimension as Wn, SearchFailureReason as Wt, scoreRedTeamOutput as X, ServedCrossFamilyError as Xn, SearchLedgerReplay as Xt, redTeamReport as Y, ModelSubstitutionError as Yn, SearchLedgerHash as Yt, CanaryAlert as Z, ServedModelCheck as Zn, SearchLedgerTrustedHeadMode as Zt, loopProvenanceArgsFromResult as _, GepaCandidatePopulationSummary as _n, costFromLedgerSummary as _t, EmitLoopProvenanceArgs as a, SearchPlannedTask as an, AssertCrossFamilyOptions as ar, OptimizationMethod as at, provenanceSpansPath as b, ExternalOptimizerExecutionSummary as bn, SearchHistoryAuditSummary as bt, LoopProvenanceBackend as c, SearchSurfaceEvidence as cn, assertCrossFamily as cr, OptimizationMethodPairwise as ct, LoopProvenanceOptimizationMethod as d, SearchTaskOutcome as dn, OptimizationMethodRunOptions as dt, SearchOperationRecordedEvent as en, assertCrossFamilyServed as er, CanaryOptions as et, LoopProvenanceRecord as f, SearchTokenAccounting as fn, OptimizationMethodScore as ft, emitLoopProvenance as g, GepaCandidatePopulationCandidate as gn, compareOptimizationMethods as gt, canonicalDigest as h, GepaCandidatePopulationArtifact as hn, combineComparisonCosts as ht, BuildLoopProvenanceArgs as i, SearchPlannedOperation as in, servedModelAcceptable as ir, ComparisonCost as it, CrowdedFrontierParentOptions as j, RunCampaignOptions as jn, FileSearchLedger as jt, RunEvalOptions as k, CampaignCellFailureReceipt as kn, searchHistoryCoverageRow as kt, LoopProvenanceCandidate as l, SearchSurfaceKind as ln, judgeFamily as lr, OptimizationMethodProvenance as lt, campaignMeasurementDigest as m, validateSearchLedgerEvent as mn, OptimizationTokenUsage as mt, isTransientTransportFailure as n, SearchPlanExtendedEvent as nn, assertServedModels as nr, runCanaries as nt, EmitLoopProvenanceResult as o, SearchSourceRef as on, CrossFamilyError as or, OptimizationMethodComparison as ot, buildLoopProvenanceRecord as p, openSearchLedger as pn, OptimizationPackageSource as pt, RedTeamReport as q, AssertCrossFamilyServedOptions as qn, SearchLedgerEntry as qt, transientDispatchFailure as r, SearchPlannedEvent as rn, checkServedModel as rr, CompareOptimizationMethodsOptions as rt, LoopProvenanceArgsFromResult as s, SearchSurfaceEffect as sn, JudgeFamily as sr, OptimizationMethodInput as st, TransientFailureOptions as t, SearchPlan as tn, assertServedModel as tr, CanaryReport as tt, LoopProvenanceEvidence as u, SearchTaskAttemptedEvent as un, OptimizationMethodResult as ut, loopProvenanceSpans as v, GepaCandidateSelectionScore as vn, optimizationTokenUsageFromSummary as vt, runImprovementLoop as w, readExternalOptimizerObservationArtifact as wn, SearchHistoryReceipt as wt, verifyLoopProvenanceRecord as x, ExternalOptimizerObservationArtifact as xn, SearchHistoryCoverage as xt, provenanceRecordPath as y, readGepaCandidatePopulationArtifact as yn, CreateSearchHistoryReceiptInput as yt, SearchRecorder as z, fsCampaignStorage as zn, SearchCandidateSlot as zt };
2109
- //# sourceMappingURL=transient-failure-DKF5Mofa.d.ts.map
2108
+ export { readExternalOptimizerObservationArtifact as $, ServedModelVerdict as $n, SearchLedgerAppendResult as $t, CanaryOptions as A, runEval as An, SearchHistoryPolicy as At, OptimizationMethodResult as B, CacheRead as Bn, SearchAccountingAudit as Bt, RedTeamReport as C, ParentSelectionContext as Cn, GepaCandidatePopulationSummary as Ct, CanaryAlert as D, OpenAutoPrResult as Dn, SearchHistoryAuditSummary as Dt, scoreRedTeamOutput as E, OpenAutoPrOptions as En, CreateSearchHistoryReceiptInput as Et, OptimizationMethod as F, CampaignRunPlan as Fn, createSearchHistoryReceipt as Ft, combineComparisonCosts as G, fsCampaignStorage as Gn, SearchCandidateRegisteredEvent as Gt, OptimizationMethodScore as H, cellCachePath as Hn, SearchAttemptAccounting as Ht, OptimizationMethodComparison as I, CampaignRunPlanCell as In, searchHistoryCoverageRow as It, optimizationTokenUsageFromSummary as J, AssertServedModelOptions as Jn, SearchCandidateSurface as Jt, compareOptimizationMethods as K, inMemoryCampaignStorage as Kn, SearchCandidateSlot as Kt, OptimizationMethodInput as L, PlanCampaignRunOptions as Ln, verifySearchHistoryReceipt as Lt, runCanaries as M, CampaignCellRetryPolicy as Mn, SearchHistoryRequiredError as Mt, CompareOptimizationMethodsOptions as N, RunCampaignOptions as Nn, assertCompleteSearchHistory as Nt, CanaryEvaluation as O, openAutoPr as On, SearchHistoryCoverage as Ot, ComparisonCost as P, runCampaign as Pn, assertSearchHistoryMatchesReplay as Pt, ExternalOptimizerSubmittedCandidate as Q, ServedModelPolicy as Qn, SearchLedger as Qt, OptimizationMethodPairwise as R, planCampaignRun as Rn, FileSearchLedger as Rt, RedTeamFinding as S, CrowdedFrontierParentOptions as Sn, GepaCandidatePopulationCandidate as St, redTeamReport as T, crowdedFrontierParent as Tn, readGepaCandidatePopulationArtifact as Tt, OptimizationPackageSource as U, CampaignStorage as Un, SearchCandidateDecidedEvent as Ut, OptimizationMethodRunOptions as V, readCachedCell as Vn, SearchArtifactRef as Vt, OptimizationTokenUsage as W, createRunCostLedger as Wn, SearchCandidateLineage as Wt, ExternalOptimizerObservationArtifact as X, ServedCrossFamilyError as Xn, SearchCostAccounting as Xt, ExternalOptimizerExecutionSummary as Y, ModelSubstitutionError as Yn, SearchCompletedEvent as Yt, ExternalOptimizerObservationSummary as Z, ServedModelCheck as Zn, SearchFailureReason as Zt, provenanceSpansPath as _, SearchTaskAttemptedEvent as _n, SearchRecorder as _t, LoopProvenanceBackend as a, SearchModelIdentity as an, AssertCrossFamilyOptions as ar, transientDispatchFailure as at, RedTeamCase as b, openSearchLedger as bn, recordCandidatePopulationSearch as bt, LoopProvenanceOptimizationMethod as c, SearchPlan as cn, assertCrossFamily as cr, runImprovementLoop as ct, campaignMeasurementDigest as d, SearchPlannedOperation as dn, RunOptimizationResult as dt, SearchLedgerEntry as en, assertCrossFamilyServed as er, LlmJudgeDimension as et, canonicalDigest as f, SearchPlannedTask as fn, runOptimization as ft, provenanceRecordPath as g, SearchSurfaceKind as gn, SearchLedgerBinding as gt, loopProvenanceSpans as h, SearchSurfaceEvidence as hn, SearchExecutionIdentity as ht, LoopProvenanceArgsFromResult as i, SearchLedgerTrustedHeadMode as in, servedModelAcceptable as ir, isTransientTransportFailure as it, CanaryReport as j, CampaignCellFailureReceipt as jn, SearchHistoryReceipt as jt, CanaryKind as k, RunEvalOptions as kn, SearchHistoryCoverageRow as kt, LoopProvenanceRecord as l, SearchPlanExtendedEvent as ln, judgeFamily as lr, PremeasuredOptimizationBaseline as lt, loopProvenanceArgsFromResult as m, SearchSurfaceEffect as mn, ProposedSearchCandidate as mt, EmitLoopProvenanceArgs as n, SearchLedgerHash as nn, assertServedModels as nr, llmJudge as nt, LoopProvenanceCandidate as o, SearchOperationKind as on, CrossFamilyError as or, RunImprovementLoopOptions as ot, emitLoopProvenance as p, SearchSourceRef as pn, MeasuredSearchCandidate as pt, costFromLedgerSummary as q, AssertCrossFamilyServedOptions as qn, SearchCandidateSlotClosedEvent as qt, EmitLoopProvenanceResult as r, SearchLedgerReplay as rn, checkServedModel as rr, TransientFailureOptions as rt, LoopProvenanceEvidence as s, SearchOperationRecordedEvent as sn, JudgeFamily as sr, RunImprovementLoopResult as st, BuildLoopProvenanceArgs as t, SearchLedgerEvent as tn, assertServedModel as tr, LlmJudgeOptions as tt, buildLoopProvenanceRecord as u, SearchPlannedEvent as un, RunOptimizationOptions as ut, verifyLoopProvenanceRecord as v, SearchTaskOutcome as vn, SearchRecorderOptions as vt, redTeamDataset as w, ParentSelector as wn, GepaCandidateSelectionScore as wt, RedTeamCategory as x, validateSearchLedgerEvent as xn, GepaCandidatePopulationArtifact as xt, DEFAULT_RED_TEAM_CORPUS as y, SearchTokenAccounting as yn, SearchRunIdentity as yt, OptimizationMethodProvenance as z, CacheIssueReason as zn, OpenSearchLedgerOptions as zt };
2109
+ //# sourceMappingURL=provenance-CIRUardl.d.ts.map