@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,2244 @@
1
+ import { t as DefaultVerdict } from "./verdict-Dps8_okt.js";
2
+ import { c as ValidationError, t as AgentEvalError } from "./errors-CEk209JS.js";
3
+ import { a as RunRecord, s as RunSplitTag } from "./run-record-CnZu_gjl.js";
4
+ import { l as RawProviderSink } from "./raw-provider-sink-BU29Sh8h.js";
5
+ import { c as CostLedgerHandle, p as CostReceipt } from "./cost-ledger-Dye6jCgg.js";
6
+ import { A as ChatClient } from "./types-DGsxbAEd.js";
7
+ import { c as EProcessState, d as PairedBootstrapOptions, f as PairedBootstrapResult, h as RiskDifferenceResult, u as McNemarResult } from "./statistics-Cmj6nynr.js";
8
+ import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, b as GenerationRecord, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface, w as JudgeScore } from "./types-k9tZGKUg.js";
9
+ import { rn as PlanCampaignRunOptions, sn as CampaignStorage, tn as CampaignRunPlan } from "./skillopt-optimization-method-B7wX7XkF.js";
10
+ import { n as AnalyzeTracesOptions, r as AnalyzeTracesResult, t as AnalyzeTracesInput } from "./analyst-BkTS3C58.js";
11
+ import { o as LedgerHash } from "./index-BAvgST_9.js";
12
+ import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
13
+ //#region src/campaign/analyst-surface.d.ts
14
+ /**
15
+ * A labeled trace scenario: a FIXED trace corpus plus the failure modes a
16
+ * competent analyst MUST surface from it. The labels are ground truth — the
17
+ * objective failures that actually occurred — which is what makes optimizing
18
+ * the analyst prompt against them meaningful rather than circular.
19
+ */
20
+ interface AnalystScenario extends Scenario {
21
+ kind: 'analyst-surface';
22
+ /** OTLP-JSONL path or an in-memory store of the traces to analyze. */
23
+ source: AnalyzeTracesOptions['source'];
24
+ /** The domain question handed to the analyst (framing lives here, not in
25
+ * the surface under optimization). */
26
+ question: string;
27
+ /**
28
+ * Ground-truth failure modes a good analyst must identify. A finding "hits"
29
+ * a mode when it contains ANY of the mode's case-insensitive cues. Derive
30
+ * these from objective signal (failed task + which step broke), never from
31
+ * the analyst's own prior output.
32
+ */
33
+ expectedFailureModes: Array<{
34
+ id: string;
35
+ cues: string[];
36
+ }>;
37
+ /**
38
+ * Cues that mark a finding as HALLUCINATED / out-of-scope for this corpus —
39
+ * naming a tool, error, or failure that did not occur. Presence penalizes
40
+ * precision. Optional; omit to score recall only.
41
+ */
42
+ forbiddenCues?: string[];
43
+ }
44
+ /** The analyst's output for one scenario — the artifact the judge scores. */
45
+ interface AnalystArtifact {
46
+ answer: string;
47
+ findings: string[];
48
+ /** The hardcoded-prompt version the analyst reported (provenance only; the
49
+ * optimized surface overrides the actual prompt text used). */
50
+ actorPromptVersion: string;
51
+ }
52
+ interface BuildAnalystSurfaceDispatchOptions {
53
+ /**
54
+ * Everything `analyzeTraces` needs EXCEPT `actorDescription` (supplied by the
55
+ * surface under optimization) and `source` (supplied by the scenario). `ai`
56
+ * (the AxAIService) is required for a live run.
57
+ */
58
+ analystOptions: Omit<AnalyzeTracesOptions, 'actorDescription' | 'source'>;
59
+ /** Test seam: defaults to the real `analyzeTraces`. */
60
+ analyze?: (input: AnalyzeTracesInput, options: AnalyzeTracesOptions) => Promise<AnalyzeTracesResult>;
61
+ }
62
+ /**
63
+ * Build the `dispatchWithSurface(surface, scenario, ctx)` the improvement loop
64
+ * calls: run the analyst with `surface` as its actorDescription over the
65
+ * scenario's trace corpus and return its findings.
66
+ */
67
+ declare function buildAnalystSurfaceDispatch(opts: BuildAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: AnalystScenario, ctx: DispatchContext) => Promise<AnalystArtifact>;
68
+ interface FailureModeRecallJudgeOptions {
69
+ /** Weight on recall when precision is also scored (forbiddenCues present).
70
+ * Default 0.5 (equal). Recall-only when no forbiddenCues exist. */
71
+ recallWeight?: number;
72
+ }
73
+ /**
74
+ * Deterministic, ground-truth judge for analyst findings. Composite =
75
+ * recall of the scenario's `expectedFailureModes` (optionally blended with a
76
+ * precision term that penalizes findings tripping `forbiddenCues`). No LLM —
77
+ * the score is a function of the labels, so the analyst prompt is optimized
78
+ * toward surfacing real failures, not toward a judge it can flatter.
79
+ */
80
+ declare function failureModeRecallJudge(opts?: FailureModeRecallJudgeOptions): JudgeConfig<AnalystArtifact, AnalystScenario>;
81
+ //#endregion
82
+ //#region src/paired-arms.d.ts
83
+ /** One arm observation of one work item. Structural on purpose: callers
84
+ * project their own record type (e.g. a `RunRecord`) into this shape. */
85
+ interface PairedArmRow {
86
+ /** Matching key — rows sharing a `pairKey` across both arms form pairs
87
+ * (typically the task/scenario/seed identity). */
88
+ pairKey: string;
89
+ /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
90
+ * every row of a `pairKey` that has more than one rep in either arm; reps
91
+ * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
92
+ * content. Optional when each arm has at most one rep of the item. */
93
+ repKey?: string;
94
+ /** Arm label this row was produced under. */
95
+ arm: string;
96
+ /** Binary outcome; omit when the comparison has no pass/fail notion. */
97
+ pass?: boolean;
98
+ /** Named numeric measurements (score, cost, latency, …). */
99
+ metrics?: Record<string, number>;
100
+ }
101
+ interface PairArmsOptions {
102
+ /** Arm treated as the control side of every pair. */
103
+ baselineArm: string;
104
+ /** Arm treated as the treatment side of every pair. */
105
+ treatmentArm: string;
106
+ }
107
+ /** One matched (baseline, treatment) observation of the same work item. */
108
+ interface MatchedPair {
109
+ pairKey: string;
110
+ /** 0-based position of this pair within its `pairKey`, ordered by sorted
111
+ * `repKey` (always 0 for a single-rep item). The rep identity itself is on
112
+ * the rows (`baseline.repKey` / `treatment.repKey`). */
113
+ repIndex: number;
114
+ baseline: PairedArmRow;
115
+ treatment: PairedArmRow;
116
+ }
117
+ interface PairArmsResult {
118
+ /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
119
+ pairs: MatchedPair[];
120
+ /** Baseline rows left without a treatment counterpart — reported, never
121
+ * silently dropped. */
122
+ unpairedBaseline: PairedArmRow[];
123
+ /** Treatment rows left without a baseline counterpart. */
124
+ unpairedTreatment: PairedArmRow[];
125
+ }
126
+ /**
127
+ * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
128
+ *
129
+ * A `pairKey` with at most one row per arm pairs directly, no `repKey`
130
+ * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
131
+ * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
132
+ * match — pairing is keyed purely on row identity, never on outcome content
133
+ * (outcome-keyed matching deflates discordant counts and biases McNemar), and
134
+ * is therefore independent of input order. Reps whose `repKey` has no
135
+ * counterpart in the other arm, and items present in only one arm, land in
136
+ * the unpaired lists — reported, never truncated.
137
+ *
138
+ * Fail-loud: throws when either named arm has zero rows (an unknown arm
139
+ * name would otherwise read as "everything unpaired"), when the two arm
140
+ * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
141
+ * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
142
+ * ambiguous).
143
+ */
144
+ declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
145
+ /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
146
+ interface PairedCorrectness {
147
+ /** Discordant pairs where the treatment passed and the baseline failed. */
148
+ b10: number;
149
+ /** Discordant pairs where the baseline passed and the treatment failed. */
150
+ b01: number;
151
+ /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
152
+ mcnemar: McNemarResult;
153
+ /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
154
+ riskDifference: RiskDifferenceResult;
155
+ }
156
+ /** Paired delta summary for one named metric (delta = treatment − baseline). */
157
+ interface PairedMetricDelta {
158
+ name: string;
159
+ /** Pairs where BOTH sides carry a finite value for this metric. */
160
+ n: number;
161
+ /** Pairs where at least one side does not carry the metric. */
162
+ nMissing: number;
163
+ /** Median paired delta, or null when `n === 0`. */
164
+ medianDelta: number | null;
165
+ /** Mean paired delta, or null when `n === 0`. */
166
+ meanDelta: number | null;
167
+ /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
168
+ * `n === 0` — a zero-width [0, 0] interval on no data would read as a
169
+ * measured tight null. */
170
+ bootstrapCi: PairedBootstrapResult | null;
171
+ /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
172
+ wilcoxon: {
173
+ w: number;
174
+ p: number;
175
+ } | null;
176
+ }
177
+ interface ComparePairedArmsOptions extends PairArmsOptions {
178
+ /** Metrics to compare. Default: every metric name observed on any matched
179
+ * pair, sorted. A name that appears on no pair is still reported (with
180
+ * `n = 0`) so a misspelled metric is visible instead of vanishing. */
181
+ metricNames?: string[];
182
+ /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
183
+ bootstrap?: PairedBootstrapOptions;
184
+ }
185
+ interface PairedArmsComparison {
186
+ nPairs: number;
187
+ nUnpairedBaseline: number;
188
+ nUnpairedTreatment: number;
189
+ /** null when no matched pair carries `pass` on both sides — a pass/fail
190
+ * verdict over rows that never measured pass/fail would be fabricated. */
191
+ correctness: PairedCorrectness | null;
192
+ metricDeltas: PairedMetricDelta[];
193
+ }
194
+ /**
195
+ * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
196
+ * the paired estimators from `statistics` over the matched pairs.
197
+ *
198
+ * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
199
+ * is that subset's size); each metric uses only the pairs where both sides
200
+ * carry a finite value for it, with the remainder counted in `nMissing`.
201
+ * Deltas are treatment − baseline throughout.
202
+ *
203
+ * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
204
+ * non-finite metric value — silently treating corrupt telemetry as "metric
205
+ * absent" would misreport it as missing coverage.
206
+ */
207
+ declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
208
+ interface MatchedRunRecordPair {
209
+ pairKey: string;
210
+ repKey: string;
211
+ baseline: RunRecord;
212
+ treatment: RunRecord;
213
+ }
214
+ interface PairRunRecordsResult {
215
+ pairs: MatchedRunRecordPair[];
216
+ unpairedBaseline: RunRecord[];
217
+ unpairedTreatment: RunRecord[];
218
+ }
219
+ /**
220
+ * Pair two RunRecord arms by the identity of the evaluated work:
221
+ * `(experimentId, scenarioId, seed)`.
222
+ *
223
+ * Falling back to array order, candidate id, or experiment id can compare
224
+ * different tasks and fabricate lift. Duplicate identities throw.
225
+ */
226
+ declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
227
+ //#endregion
228
+ //#region src/campaign/cross-surface-types.d.ts
229
+ /** Whether one candidate attempt produced a usable executable outcome. */
230
+ type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
231
+ /** One independently proposed change on one caller-defined surface. */
232
+ interface CrossSurfaceComponent {
233
+ componentId: string;
234
+ surfaceId: string;
235
+ /** Explicitly controls whether this component may anchor the best-single arm. */
236
+ bestSingleEligible: boolean;
237
+ }
238
+ /** Immutable identity for a single candidate or a materialized composition. */
239
+ interface CrossSurfaceCandidate {
240
+ candidateId: string;
241
+ componentIds: string[];
242
+ contentHash: string;
243
+ artifactBytes: number;
244
+ }
245
+ /** Per-component trace evidence captured during one task attempt. */
246
+ interface CrossSurfaceComponentEvidence {
247
+ componentId: string;
248
+ /** null means the trace could not establish whether the component fired. */
249
+ fired: boolean | null;
250
+ /** null means the trace could not establish whether the component changed behavior. */
251
+ effectObserved: boolean | null;
252
+ }
253
+ /**
254
+ * Canonical per-task input row. Consumers may extend this interface with
255
+ * receipt, trace, retry, or failure details; the report preserves the original
256
+ * row object rather than projecting those details away.
257
+ */
258
+ interface CrossSurfaceTaskRow {
259
+ taskId: string;
260
+ candidateId: string;
261
+ /** Repeated here so every persisted row remains self-describing. */
262
+ componentIds: string[];
263
+ completeness: CrossSurfaceAttemptCompleteness;
264
+ pass: boolean | null;
265
+ score: number | null;
266
+ /**
267
+ * Per-attempt deployment measurements. Every declared metric must have a
268
+ * known, non-negative value. Proposal, analysis, and selection spend belongs
269
+ * in the search ledger rather than being spread across task cells.
270
+ */
271
+ cost: Record<string, number | null>;
272
+ componentEvidence: CrossSurfaceComponentEvidence[];
273
+ /** Required for missing or invalid attempts; forbidden for complete attempts. */
274
+ rejectReason: string | null;
275
+ }
276
+ interface CrossSurfaceBootstrapPolicy {
277
+ seed: number;
278
+ resamples: number;
279
+ confidence: number;
280
+ }
281
+ /** Predeclared candidate eligibility and composition policy. */
282
+ interface CrossSurfaceSelectionPolicy {
283
+ minimumFiringTasks: number;
284
+ minimumEffectTasks: number;
285
+ requireObservedFiring: boolean;
286
+ requireObservedEffect: boolean;
287
+ /** Only named metrics are constrained; all declared metrics are still reported. */
288
+ maximumMedianCostRatioToBaseline: Record<string, number>;
289
+ /** A smaller terminal bundle is reported but cannot become the selected arm. */
290
+ minimumBundleComponents: number;
291
+ }
292
+ interface AnalyzeCrossSurfaceInteractionsInput<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
293
+ components: readonly CrossSurfaceComponent[];
294
+ candidates: readonly CrossSurfaceCandidate[];
295
+ rows: readonly TRow[];
296
+ baselineCandidateId: string;
297
+ /** Exact shared task axis and its canonical output order. */
298
+ taskOrder: readonly string[];
299
+ /** Canonical materialization order for component sets and the naive stack. */
300
+ componentOrder: readonly string[];
301
+ /** Final deterministic tie-break; lower index wins. */
302
+ candidateOrder: readonly string[];
303
+ /** Declares every cost key and the order used for cost tie-breaks. */
304
+ costMetricOrder: readonly string[];
305
+ bootstrap: CrossSurfaceBootstrapPolicy;
306
+ selection: CrossSurfaceSelectionPolicy;
307
+ }
308
+ interface CrossSurfaceDistribution {
309
+ n: number;
310
+ min: number;
311
+ median: number;
312
+ mean: number;
313
+ max: number;
314
+ total: number;
315
+ }
316
+ interface CrossSurfaceEvidenceBreakdown {
317
+ componentId: string;
318
+ observedTaskIds: string[];
319
+ notObservedTaskIds: string[];
320
+ unobservedTaskIds: string[];
321
+ }
322
+ interface CrossSurfaceCandidateEvidence {
323
+ byComponent: CrossSurfaceEvidenceBreakdown[];
324
+ allObservedTaskIds: string[];
325
+ someObservedTaskIds: string[];
326
+ noneObservedTaskIds: string[];
327
+ unobservedTaskIds: string[];
328
+ }
329
+ type CrossSurfaceIneligibilityReason = 'missing_attempt' | 'invalid_attempt' | 'baseline_outcome_missing' | 'benefit_not_greater_than_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
330
+ interface CrossSurfaceEligibility {
331
+ eligible: boolean;
332
+ reasons: CrossSurfaceIneligibilityReason[];
333
+ }
334
+ interface CrossSurfaceCandidateOutcome {
335
+ resolvedTaskIds: string[];
336
+ failedTaskIds: string[];
337
+ missingTaskIds: string[];
338
+ invalidTaskIds: string[];
339
+ benefitTaskIds: string[];
340
+ regressionTaskIds: string[];
341
+ comparisonMissingTaskIds: string[];
342
+ netBenefit: number;
343
+ }
344
+ interface CrossSurfaceCandidateSummary {
345
+ candidate: CrossSurfaceCandidate;
346
+ outcome: CrossSurfaceCandidateOutcome;
347
+ score: CrossSurfaceDistribution | null;
348
+ costs: Record<string, CrossSurfaceDistribution>;
349
+ firing: CrossSurfaceCandidateEvidence;
350
+ effect: CrossSurfaceCandidateEvidence;
351
+ /** Reuses the package's paired McNemar/risk-difference/bootstrap statistics. */
352
+ comparisonToBaseline: PairedArmsComparison | null;
353
+ /** null only for the fixed baseline. */
354
+ eligibility: CrossSurfaceEligibility | null;
355
+ }
356
+ interface CrossSurfaceRelativeCost {
357
+ treatmentMedian: number;
358
+ comparatorMedian: number;
359
+ medianDelta: number;
360
+ /** null when the comparator median is zero but the treatment median is not. */
361
+ medianRatio: number | null;
362
+ }
363
+ interface CrossSurfaceCandidateComparison {
364
+ comparatorCandidateId: string;
365
+ treatmentCandidateId: string;
366
+ winsTaskIds: string[];
367
+ regressionTaskIds: string[];
368
+ missingTaskIds: string[];
369
+ paired: PairedArmsComparison;
370
+ relativeCost: Record<string, CrossSurfaceRelativeCost>;
371
+ }
372
+ interface CrossSurfacePairEvidence {
373
+ bothTaskIds: string[];
374
+ leftOnlyTaskIds: string[];
375
+ rightOnlyTaskIds: string[];
376
+ neitherTaskIds: string[];
377
+ unobservedTaskIds: string[];
378
+ }
379
+ interface CrossSurfaceInteractionTask {
380
+ taskId: string;
381
+ /** Composition minus the additive expectation from the baseline and singles. */
382
+ passInteraction: number | null;
383
+ scoreInteraction: number | null;
384
+ }
385
+ interface CrossSurfaceInteractionEffect {
386
+ perTask: CrossSurfaceInteractionTask[];
387
+ n: number;
388
+ nMissing: number;
389
+ meanPassInteraction: number | null;
390
+ meanScoreInteraction: number | null;
391
+ passBootstrap: PairedBootstrapResult | null;
392
+ scoreBootstrap: PairedBootstrapResult | null;
393
+ }
394
+ type CrossSurfacePairIncompatibilityReason = 'constituent_not_ready' | 'pair_incomplete' | 'baseline_regression' | 'interference' | 'no_incremental_resolution' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
395
+ interface CrossSurfacePairCompatibility {
396
+ compatible: boolean;
397
+ reasons: CrossSurfacePairIncompatibilityReason[];
398
+ betterSingleCandidateId: string;
399
+ }
400
+ interface CrossSurfacePairwiseEntry {
401
+ componentIds: [string, string];
402
+ singleCandidateIds: [string, string];
403
+ compositionCandidateId: string;
404
+ benefitTaskIds: string[];
405
+ regressionTaskIds: string[];
406
+ synergyTaskIds: string[];
407
+ interferenceTaskIds: string[];
408
+ incrementalVsConstituents: [CrossSurfaceCandidateComparison, CrossSurfaceCandidateComparison];
409
+ relativeCostToBaseline: Record<string, CrossSurfaceRelativeCost>;
410
+ firing: CrossSurfacePairEvidence;
411
+ effect: CrossSurfacePairEvidence;
412
+ interaction: CrossSurfaceInteractionEffect;
413
+ compatibility: CrossSurfacePairCompatibility;
414
+ }
415
+ interface CrossSurfaceRankedSingle {
416
+ rank: number;
417
+ candidateId: string;
418
+ componentId: string;
419
+ }
420
+ interface CrossSurfaceBestSingleSelection {
421
+ candidateId: string;
422
+ componentId: string;
423
+ ranking: CrossSurfaceRankedSingle[];
424
+ }
425
+ interface CrossSurfaceNaiveStackSelection {
426
+ /** Every individually eligible single, stacked in canonical component order. */
427
+ candidateId: string;
428
+ componentIds: string[];
429
+ }
430
+ type CrossSurfaceAdditionRejectionReason = 'pair_incompatible' | 'full_bundle_not_evaluated' | 'bundle_incomplete' | 'baseline_regression' | 'no_incremental_resolution' | 'incremental_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
431
+ interface CrossSurfaceAdditionDecision {
432
+ additionCandidateId: string;
433
+ additionComponentId: string;
434
+ bundleCandidateId: string | null;
435
+ incrementalResolutionTaskIds: string[];
436
+ incrementalRegressionTaskIds: string[];
437
+ incrementalMedianCost: Record<string, number> | null;
438
+ eligible: boolean;
439
+ selected: boolean;
440
+ reasons: CrossSurfaceAdditionRejectionReason[];
441
+ }
442
+ interface CrossSurfaceCompositionStep {
443
+ fromCandidateId: string;
444
+ retainedComponentIds: string[];
445
+ considered: CrossSurfaceAdditionDecision[];
446
+ selectedCandidateId: string | null;
447
+ }
448
+ /** One deterministic growth path starting from a compatible two-surface seed. */
449
+ interface CrossSurfaceInteractionPath {
450
+ seedCandidateId: string;
451
+ terminalCandidateId: string;
452
+ terminalComponentIds: string[];
453
+ qualified: boolean;
454
+ steps: CrossSurfaceCompositionStep[];
455
+ }
456
+ interface CrossSurfaceInteractionAwareSelection {
457
+ /** Compatible pair that seeded the selected deterministic growth path. */
458
+ seedCandidateId: string;
459
+ /** Candidate reached by the winning path, even if the minimum size is not met. */
460
+ terminalCandidateId: string;
461
+ terminalComponentIds: string[];
462
+ /** null when no path produced a qualifying multi-component bundle. */
463
+ selectedCandidateId: string | null;
464
+ qualified: boolean;
465
+ /** Every compatible pair seed is retained so seed choice cannot hide an interaction. */
466
+ evaluatedPaths: CrossSurfaceInteractionPath[];
467
+ /** Convenience alias for the winning path's steps. */
468
+ steps: CrossSurfaceCompositionStep[];
469
+ }
470
+ interface CrossSurfaceSelections {
471
+ bestSingle: CrossSurfaceBestSingleSelection | null;
472
+ naiveStack: CrossSurfaceNaiveStackSelection | null;
473
+ interactionAware: CrossSurfaceInteractionAwareSelection | null;
474
+ }
475
+ interface CrossSurfaceInteractionReport<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
476
+ taskIds: string[];
477
+ componentIds: string[];
478
+ candidateIds: string[];
479
+ costMetrics: string[];
480
+ /** Canonical candidate × task order; no input row is dropped. */
481
+ rows: TRow[];
482
+ missingAttempts: TRow[];
483
+ invalidAttempts: TRow[];
484
+ candidates: CrossSurfaceCandidateSummary[];
485
+ pairwise: CrossSurfacePairwiseEntry[];
486
+ selections: CrossSurfaceSelections;
487
+ }
488
+ //#endregion
489
+ //#region src/campaign/cross-surface-interaction.d.ts
490
+ /**
491
+ * Build the complete cross-surface evidence matrix and derive all three frozen
492
+ * candidates. The task/candidate/component orders are part of the input so
493
+ * neither insertion order nor an after-the-fact tie-break can change a result.
494
+ */
495
+ declare function analyzeCrossSurfaceInteractions<TRow extends CrossSurfaceTaskRow>(input: AnalyzeCrossSurfaceInteractionsInput<TRow>): CrossSurfaceInteractionReport<TRow>;
496
+ //#endregion
497
+ //#region src/campaign/fixtures.d.ts
498
+ type EvalFixtureValidationMode = 'vitest' | 'none';
499
+ interface EvalFixtureFile {
500
+ path: string;
501
+ sha256: string;
502
+ bytes: number;
503
+ }
504
+ interface EvalFixture {
505
+ name: string;
506
+ path: string;
507
+ promptPath: string;
508
+ evalPath?: string;
509
+ packageJsonPath?: string;
510
+ prompt: string;
511
+ files: EvalFixtureFile[];
512
+ fingerprint: string;
513
+ }
514
+ interface EvalFixtureScenario extends Scenario {
515
+ kind: 'eval-fixture';
516
+ fixtureName: string;
517
+ fixturePath: string;
518
+ promptPath: string;
519
+ evalPath?: string;
520
+ packageJsonPath?: string;
521
+ prompt: string;
522
+ fingerprint: string;
523
+ }
524
+ interface EvalFixtureLoadOptions {
525
+ /** `vitest` requires EVAL.ts/EVAL.tsx and package.json type=module. `none` only requires PROMPT.md. */
526
+ validation?: EvalFixtureValidationMode;
527
+ /** Extra caller-owned knobs that affect fixture behavior, folded into the fingerprint. */
528
+ fingerprintConfig?: unknown;
529
+ }
530
+ interface LoadEvalFixtureScenariosOptions extends EvalFixtureLoadOptions {
531
+ names?: string[];
532
+ }
533
+ interface PlanEvalFixtureRunOptions<TArtifact = unknown> extends Pick<PlanCampaignRunOptions<EvalFixtureScenario, TArtifact>, 'dispatchRef' | 'judges' | 'seed' | 'reps' | 'resumable' | 'runDir'> {
534
+ evalsDir: string;
535
+ validation?: EvalFixtureValidationMode;
536
+ fingerprintConfig?: unknown;
537
+ names?: string[];
538
+ storage?: CampaignStorage;
539
+ }
540
+ type EvalFixtureRunPlan = CampaignRunPlan & {
541
+ fixtures: Array<Pick<EvalFixtureScenario, 'fixtureName' | 'fixturePath' | 'fingerprint'>>;
542
+ };
543
+ /** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
544
+ declare function discoverEvalFixtures(evalsDir: string): string[];
545
+ /**
546
+ * Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
547
+ * `vitest` validation) and content-fingerprints the full file set for cache identity.
548
+ */
549
+ declare function loadEvalFixture(evalsDir: string, name: string, options?: EvalFixtureLoadOptions): EvalFixture;
550
+ /** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
551
+ declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFixtureScenariosOptions): EvalFixtureScenario[];
552
+ /**
553
+ * Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
554
+ * and returns the plan plus each fixture's name/path/fingerprint.
555
+ */
556
+ declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
557
+ //#endregion
558
+ //#region src/campaign/gates/neutralization-gate.d.ts
559
+ interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
560
+ scenarios: TScenario[];
561
+ /** Reject when the neutralized (content-blanked, footprint-matched) variant
562
+ * reproduces at least this fraction of the candidate's held-out lift. Default
563
+ * 0.5 — if blanking the content keeps half the lift, the content is decorative.
564
+ * Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
565
+ maxDecorativeFraction?: number;
566
+ }
567
+ /**
568
+ * Composable placebo gate: ships only when the candidate's held-out lift is NOT
569
+ * mostly reproduced by a footprint-matched neutralized variant.
570
+ */
571
+ declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
572
+ //#endregion
573
+ //#region src/pre-registration.d.ts
574
+ /**
575
+ * Pre-registered hypotheses — declare what you're testing BEFORE the
576
+ * run, check it AFTER. Prevents p-hacking, optional stopping, and the
577
+ * "we ran until it looked good" failure mode.
578
+ *
579
+ * Manifest is a plain JSON-friendly object. Sign it with a content hash
580
+ * + timestamp; the registered record becomes immutable. Post-run,
581
+ * evaluate the manifest against observed results — the library refuses
582
+ * to let you re-interpret a different metric as the declared one.
583
+ */
584
+ interface HypothesisManifest {
585
+ id: string;
586
+ /** Human prose — goes into the audit trail. */
587
+ hypothesis: string;
588
+ /** Metric the hypothesis claims to move. */
589
+ metric: string;
590
+ /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
591
+ direction: 'increase' | 'decrease';
592
+ /** Minimum effect size to count (same units as the metric). */
593
+ minEffect: number;
594
+ /** Alpha threshold. */
595
+ alpha: number;
596
+ /** Target statistical power at which sample size was pre-computed. */
597
+ power: number;
598
+ /** Declared N per arm before running. */
599
+ preRegisteredN: number;
600
+ /** ISO8601 timestamp the manifest was registered. */
601
+ registeredAt: string;
602
+ /** Optional identifiers to tie into the trace corpus. */
603
+ baselineLabel?: string;
604
+ candidateLabel?: string;
605
+ }
606
+ /**
607
+ * Identifier for the hashing scheme used to produce `contentHash`.
608
+ *
609
+ * `'sha256-content'` — sha256 hex over the canonicalized manifest with
610
+ * the `contentHash` and `algo` fields stripped. Held as a string union
611
+ * so future schemes can be added without breaking parsers; SignedManifest
612
+ * values without `algo` deserialize cleanly because the field is optional.
613
+ */
614
+ type SignedManifestAlgo = 'sha256-content';
615
+ interface SignedManifest extends HypothesisManifest {
616
+ /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
617
+ contentHash: string;
618
+ /**
619
+ * Algorithm string describing how `contentHash` was produced.
620
+ *
621
+ * Optional on the type so serialized manifests without it still parse,
622
+ * but ALWAYS populated by {@link signManifest}. Consumers that want to
623
+ * enforce a known algorithm should reject manifests where this field
624
+ * is missing or unrecognized.
625
+ */
626
+ algo?: SignedManifestAlgo;
627
+ }
628
+ interface HypothesisResult {
629
+ manifest: SignedManifest;
630
+ observedN: number;
631
+ observedEffect: number;
632
+ observedPValue: number;
633
+ /** True iff the observed effect hits the pre-declared direction with
634
+ * magnitude ≥ minEffect AND p < alpha. */
635
+ confirmed: boolean;
636
+ /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
637
+ rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
638
+ notes?: string;
639
+ }
640
+ /**
641
+ * Deterministic JSON canonicalization — sort object keys recursively.
642
+ *
643
+ * Two semantically-equal objects produce byte-identical canonicalized output;
644
+ * this is what makes a content-hash stable across encoders, key insertion
645
+ * orders, and runtime versions. Exported for any consumer that needs the same
646
+ * canonicalization guarantee outside the manifest-signing path (e.g., signing
647
+ * an artifact bundle, hashing a dataset version, etc.).
648
+ */
649
+ declare function canonicalize(v: unknown): unknown;
650
+ /**
651
+ * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
652
+ *
653
+ * The same primitive `signManifest` and `verifyManifest` are built on, exposed
654
+ * directly so consumers signing arbitrary structured content (artifact bundles,
655
+ * production packets, dataset manifests, etc.) don't have to re-derive
656
+ * canonicalize+sha256 from scratch.
657
+ *
658
+ * Stable across:
659
+ * - object key insertion order (canonicalization sorts keys recursively)
660
+ * - encoder choice (UTF-8 via TextEncoder, fixed)
661
+ * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
662
+ *
663
+ * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
664
+ * which takes a string input and returns a truncated 12-char prompt id.
665
+ * Use `hashJson` when you mean "canonicalize then hash."
666
+ *
667
+ * @example
668
+ * const hash = await hashJson({ id: '1', kind: 'spec' })
669
+ * // 'a3f1...' (64 hex chars)
670
+ */
671
+ declare function hashJson<T>(obj: T): Promise<string>;
672
+ /**
673
+ * Sign a manifest with a SHA-256 content hash.
674
+ *
675
+ * The hash covers the canonicalized manifest with the `contentHash`
676
+ * and `algo` fields stripped; this lets verifiers re-sign the rest and
677
+ * compare. Returned manifest always carries `algo: 'sha256-content'`
678
+ * so downstream consumers can identify the scheme; manifests without
679
+ * `algo` still verify because it is stripped before hashing on both sides.
680
+ */
681
+ declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
682
+ /**
683
+ * Verify that a signed manifest has not been tampered with.
684
+ *
685
+ * Strips `contentHash` and `algo` before re-signing so manifests without
686
+ * `algo` verify identically to ones that carry it.
687
+ */
688
+ declare function verifyManifest(m: SignedManifest): Promise<boolean>;
689
+ /**
690
+ * Evaluate a pre-registered hypothesis against observed results.
691
+ * Mechanical — no re-interpretation permitted.
692
+ */
693
+ declare function evaluateHypothesis(manifest: SignedManifest, observed: {
694
+ n: number;
695
+ effect: number;
696
+ pValue: number;
697
+ }): Promise<HypothesisResult>;
698
+ //#endregion
699
+ //#region src/campaign/gates/sequential.d.ts
700
+ type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
701
+ interface SequentialObservation {
702
+ decision: SequentialDecision;
703
+ /** Current e-value (the betting wealth) against H0. */
704
+ eValue: number;
705
+ /** Paired deltas consumed so far. */
706
+ n: number;
707
+ /** Names the decision basis. For 'undecided-at-maxN' it states explicitly
708
+ * that exhausting the budget is NOT evidence of no effect. */
709
+ reason: string;
710
+ }
711
+ interface SequentialPairedGateOptions {
712
+ /** Type-I budget. With `preRegistration` bound this MUST match
713
+ * `manifest.alpha` (conflict throws). Default 0.05. */
714
+ alpha?: number;
715
+ /** Minimum paired deltas before a promote may fire. The stopping rule is
716
+ * "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
717
+ * Default 5. */
718
+ minN?: number;
719
+ /** Pre-registered observation budget. Required unless `preRegistration`
720
+ * supplies it via `preRegisteredN` (conflict throws). */
721
+ maxN?: number;
722
+ /** Bet truncation forwarded to `eProcess`. Default 0.5. */
723
+ maxBet?: number;
724
+ /** Bound on |delta| in the judge's native scale; deltas are mapped to
725
+ * x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
726
+ * `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
727
+ scale?: number;
728
+ /** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
729
+ * (exchangeability guard). Default 1337. */
730
+ shuffleSeed?: number;
731
+ /** Bind the pre-registered hypothesis. Verified (content hash) at
732
+ * construction; alpha/maxN/direction/minEffect come FROM the manifest. */
733
+ preRegistration?: SignedManifest;
734
+ /** Override the gate name in reports. */
735
+ name?: string;
736
+ }
737
+ interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
738
+ /** Streaming entry point: feed one paired per-scenario delta
739
+ * (candidate − baseline, native scale). Each gate instance carries ONE
740
+ * observe-stream; `decide(ctx)` runs on its own fresh stream and never
741
+ * consumes or advances this one. 'promote' is sticky; observing past the
742
+ * pre-registered maxN throws (extending a finished stream after seeing
743
+ * the result reopens optional stopping — start a NEW pre-registered
744
+ * test). */
745
+ observe(delta: number): SequentialObservation;
746
+ /** Read-only snapshot of the observe-stream. */
747
+ state(): EProcessState & {
748
+ decision: SequentialDecision;
749
+ };
750
+ }
751
+ /**
752
+ * Anytime-valid sequential paired gate. Conforms to the existing `Gate`
753
+ * contract (`decide(ctx)` consumes candidate vs baseline judge scores via
754
+ * `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
755
+ * never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
756
+ * that score cells incrementally and want to stop mid-stream.
757
+ *
758
+ * Decision mapping onto the substrate's five-valued `GateDecision`:
759
+ * - 'promote' → 'ship'
760
+ * - 'continue' → 'need_more_work' (stream ended before maxN with
761
+ * the e-value undecided — more reps could decide)
762
+ * - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
763
+ * evidence of no effect (never a silent default)
764
+ */
765
+ declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
766
+ interface SequentialDecideOptions {
767
+ /** Type-I budget for the early-stop evidence. Default 0.05. */
768
+ alpha?: number;
769
+ /** Minimum paired deltas before a stop may fire. Default 5. */
770
+ minN?: number;
771
+ /** Bet truncation forwarded to `eProcess`. Default 0.5. */
772
+ maxBet?: number;
773
+ /** Bound on |per-scenario composite delta|. Default 1. */
774
+ scale?: number;
775
+ }
776
+ interface SequentialDecideFn {
777
+ (args: {
778
+ history: GenerationRecord[];
779
+ }): {
780
+ stop: boolean;
781
+ reason?: string;
782
+ };
783
+ /** Read-only snapshot of the accumulated e-process (observability + tests). */
784
+ state(): EProcessState;
785
+ }
786
+ /**
787
+ * `SurfaceProposer.decide` adapter — stops the optimization loop the moment
788
+ * the e-process decides the loop has produced a real improvement, instead of
789
+ * always running `maxGenerations`.
790
+ *
791
+ * Stream: for each generation g ≥ 1, the per-scenario composite deltas of
792
+ * generation g's top candidate vs the generation-0 top candidate (the
793
+ * incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
794
+ * surface improves any scenario's expected composite over the incumbent —
795
+ * under it every delta has conditional mean ≤ 0 and the e-process is valid.
796
+ * Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
797
+ * gate (which re-scores on HELD-OUT data — this adapter only spends the
798
+ * exploration budget, it never promotes).
799
+ *
800
+ * Honesty caveats: (1) the incumbent's scores are measured once and shared
801
+ * across all generations' deltas, so type-I control is exact only insofar as
802
+ * those scores approximate the incumbent's true per-scenario means (more reps
803
+ * → tighter); (2) an UNDECIDED process never stops the loop — absence of a
804
+ * crossing is NOT evidence of no effect, so the loop simply runs its normal
805
+ * course. Calling the adapter repeatedly with a growing history consumes each
806
+ * generation exactly once (re-feeding an already-seen record would double-count
807
+ * evidence).
808
+ */
809
+ declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
810
+ //#endregion
811
+ //#region src/campaign/gates/statistical-heldout.d.ts
812
+ interface PairedHoldout {
813
+ /** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
814
+ before: number[];
815
+ /** Candidate scalar per paired cell. */
816
+ after: number[];
817
+ /** The full cellIds (`scenario:rep`) that paired, in order. */
818
+ cellIds: string[];
819
+ }
820
+ /**
821
+ * Pair candidate vs baseline holdout observations by FULL cellId. `select`
822
+ * pulls the scalar from a cell's judge reports (composite, or a named
823
+ * dimension); a cell contributes the mean of `select` across its judges. Cells
824
+ * whose scenario is not in `scenarioIds`, or where `select` is undefined for
825
+ * every judge on either side, are skipped on BOTH sides so the arrays stay
826
+ * paired. Throws when the two maps disagree on which holdout cells exist — a
827
+ * load-bearing invariant: the baseline + winner holdout campaigns run the same
828
+ * scenarios with the same seed base, so their cellIds MUST align; a mismatch
829
+ * means a silent pairing bug, not a soft fallback.
830
+ */
831
+ declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
832
+ interface HeldoutSignificance {
833
+ paired: PairedHoldout;
834
+ /** The bootstrap the ship decision keys on — of the MEAN paired delta by
835
+ * default (see the tie note on `heldoutSignificance`). */
836
+ bootstrap: PairedBootstrapResult;
837
+ /** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
838
+ * scenarios are tied (both sides solve them), the median is pinned near 0
839
+ * regardless of the mean lift — comparing the two exposes tie-domination. */
840
+ medianBootstrap: PairedBootstrapResult;
841
+ /** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
842
+ * high tie fraction is WHY a median-based gate would have missed a real lift;
843
+ * it is the observability the tie fix adds. */
844
+ tieFraction: number;
845
+ /** n paired observations. */
846
+ n: number;
847
+ /** True iff n >= minProductiveRuns AND the CI lower bound clears the threshold. */
848
+ significant: boolean;
849
+ /** Set when n < minProductiveRuns — too little evidence to claim significance. */
850
+ fewRuns: boolean;
851
+ }
852
+ interface HeldoutSignificanceOptions {
853
+ deltaThreshold?: number;
854
+ minProductiveRuns?: number;
855
+ confidence?: number;
856
+ resamples?: number;
857
+ /** Fixed by default for a deterministic, reproducible gate verdict. */
858
+ seed?: number;
859
+ statistic?: 'mean' | 'median';
860
+ }
861
+ /** Significance of the held-out composite lift: ship only when the paired
862
+ * bootstrap CI lower bound on (candidate − baseline) exceeds `deltaThreshold`
863
+ * (default 0 ⇒ "confidently positive"). Below `minProductiveRuns` paired
864
+ * observations there is not enough evidence to claim significance → not
865
+ * significant (`fewRuns`). Interpret `deltaThreshold` in the judge's native
866
+ * composite scale. */
867
+ declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
868
+ interface DimensionRegression {
869
+ dimension: string;
870
+ bootstrap: PairedBootstrapResult;
871
+ /** True iff the CI lower bound on (candidate − baseline) is below −tolerance:
872
+ * the candidate may have regressed this dimension by more than tolerance. */
873
+ regressed: boolean;
874
+ tolerance: number;
875
+ n: number;
876
+ }
877
+ /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
878
+ * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
879
+ * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
880
+ declare function detectScale(values: number[]): 1 | 100;
881
+ /** Per-critical-dimension regression guard. For each dimension, pair the
882
+ * candidate vs baseline values by full cellId and bootstrap the paired delta;
883
+ * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
884
+ * — blocks if the credible worst case exceeds tolerance, which is the right
885
+ * posture for safety dimensions like `hallucination_free`). When `tolerance`
886
+ * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100. */
887
+ declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
888
+ tolerance?: number;
889
+ confidence?: number;
890
+ resamples?: number;
891
+ seed?: number;
892
+ }): DimensionRegression[];
893
+ //#endregion
894
+ //#region src/campaign/grounded-reflection.d.ts
895
+ /**
896
+ * Evidence grounding for reflective optimizers (GEPA-style revise loops).
897
+ *
898
+ * Two failure modes recur when an LLM revises an artifact from raw rollout
899
+ * traces (first measured in agent-lab R358, where naive reflection REGRESSED
900
+ * the score 0.375 -> 0.125 before these helpers fixed it):
901
+ *
902
+ * 1. The environment often hides WHY a rollout failed - a tool call can
903
+ * succeed while an invisible downstream check fails - so the reviser
904
+ * cannot see the cause in the transcript. The only reliable signal is the
905
+ * field-level difference between what passing and failing rollouts did.
906
+ * `rolloutArgumentDiff` computes that difference deterministically so the
907
+ * reviser is handed the diff instead of being trusted to derive it.
908
+ *
909
+ * 2. Revisers invent plausible-but-wrong literal values ("use 'new'",
910
+ * "use 'sent'") that no passing rollout ever used, turning every rollout
911
+ * into a failure. `classifyUngroundedLiterals` mechanically detects them,
912
+ * separating HARMFUL literals (ones failing rollouts actually used -
913
+ * proven damage) from benign illustrations (e.g. a name example like
914
+ * 'Doe'), so callers can hard-reject the former and merely log the latter.
915
+ * Rejecting every ungrounded quoted word is too blunt: it killed a run
916
+ * over a surname illustration before the severity split existed.
917
+ *
918
+ * Pure data in, data out: no LLM calls, no filesystem, no domain knowledge.
919
+ */
920
+ /** One tool/action call observed in a rollout: a name plus its arguments. */
921
+ interface RolloutCall {
922
+ readonly name: string;
923
+ readonly args: Readonly<Record<string, unknown>>;
924
+ }
925
+ /** A scored rollout: its calls plus the scalar outcome used to split pass/fail. */
926
+ interface ScoredRollout {
927
+ /** Caller-meaningful identifier (task id, cell id) used only for reporting. */
928
+ readonly id: string;
929
+ /** Scalar outcome in [0, 1]; `passThreshold` splits passing from failing. */
930
+ readonly score: number;
931
+ readonly calls: readonly RolloutCall[];
932
+ }
933
+ interface RolloutArgumentDiffOptions {
934
+ /** Rollouts with `score >= passThreshold` count as passing. Default 1. */
935
+ readonly passThreshold?: number;
936
+ /** Max distinct values listed per field per side in the rendered text. Default 4. */
937
+ readonly maxValuesPerField?: number;
938
+ }
939
+ interface RolloutArgumentDiff {
940
+ /** Human/LLM-readable per-field diff, one line per field. */
941
+ readonly text: string;
942
+ /** Lowercased stringified argument values seen in passing rollouts. */
943
+ readonly passingValues: ReadonlySet<string>;
944
+ /** Lowercased stringified argument values seen in failing rollouts. */
945
+ readonly failingValues: ReadonlySet<string>;
946
+ }
947
+ /**
948
+ * Deterministic per-field diff of call arguments between passing and failing
949
+ * rollouts. A field set by failing rollouts but left unset by passing ones is
950
+ * the classic poison-input signature; a field whose values differ across the
951
+ * split points at the correct value. Feed `text` to the reviser verbatim.
952
+ */
953
+ declare function rolloutArgumentDiff(rollouts: readonly ScoredRollout[], opts?: RolloutArgumentDiffOptions): RolloutArgumentDiff;
954
+ interface UngroundedLiteralReport {
955
+ /** Quoted single-word literals in the text that no passing rollout used. */
956
+ readonly ungrounded: readonly string[];
957
+ /** The subset failing rollouts actually used - prescribing these is proven harmful. */
958
+ readonly harmful: readonly string[];
959
+ }
960
+ /**
961
+ * Scan revised artifact text for single-quoted single-word literals (the
962
+ * "use exactly 'new'" pattern) that appear in no passing rollout's argument
963
+ * values. Multi-word quotes pass (they are prose, not prescriptions).
964
+ * Callers should reject on `harmful` (with a bounded retry) and at most log
965
+ * `ungrounded` - see the module header for why the severities differ.
966
+ */
967
+ declare function classifyUngroundedLiterals(text: string, diff: Pick<RolloutArgumentDiff, 'passingValues' | 'failingValues'>): UngroundedLiteralReport;
968
+ //#endregion
969
+ //#region src/campaign/labeled-store/fs-adapter.d.ts
970
+ interface FsLabeledScenarioStoreOptions {
971
+ /** Root directory for JSONL files. Created if missing. */
972
+ root: string;
973
+ /** Per-source rate limit. When set, writes exceeding the cap are rejected
974
+ * with a typed error. Default: no limit. */
975
+ maxWritesPerMinutePerBucket?: number;
976
+ /** Test seam — override `Date.now()` for deterministic tests. */
977
+ now?: () => number;
978
+ }
979
+ /** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
980
+ declare class LabeledScenarioStoreError extends Error {
981
+ readonly code: string;
982
+ constructor(code: string, message: string);
983
+ }
984
+ /**
985
+ * Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
986
+ * rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
987
+ */
988
+ declare class FsLabeledScenarioStore implements LabeledScenarioStore {
989
+ private readonly options;
990
+ private readonly now;
991
+ private readonly rateLimits;
992
+ constructor(options: FsLabeledScenarioStoreOptions);
993
+ observe(write: LabeledScenarioWrite): Promise<void>;
994
+ sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
995
+ size(): Promise<{
996
+ train: number;
997
+ test: number;
998
+ bySource: Record<string, number>;
999
+ byTrust: Record<LabelTrust, number>;
1000
+ }>;
1001
+ private assertProvenance;
1002
+ private assertRateLimit;
1003
+ private toRecord;
1004
+ private pathForSource;
1005
+ }
1006
+ //#endregion
1007
+ //#region src/campaign/neutralize.d.ts
1008
+ /**
1009
+ * @module
1010
+ * Footprint-matched neutralization — the placebo control for content-vs-footprint
1011
+ * attribution in a promotion gate.
1012
+ *
1013
+ * A promoted surface can raise a held-out score two different ways:
1014
+ * 1. its CONTENT is informative (the thing we want to promote), or
1015
+ * 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
1016
+ * more authoritative-looking prompt — that the model spends attention on
1017
+ * regardless of what the bytes say.
1018
+ *
1019
+ * A held-out gate proves the candidate beat baseline; it cannot separate (1) from
1020
+ * (2). `neutralizeText` produces a variant that keeps the input's layout and
1021
+ * length while carrying ZERO information, so scoring it isolates the footprint
1022
+ * contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
1023
+ * any lift it still holds over baseline is decorative, and a candidate whose lift
1024
+ * survives neutralization is rejected however large its raw lift.
1025
+ */
1026
+ /**
1027
+ * Blank every non-whitespace character to a 1-byte filler while preserving all
1028
+ * whitespace. Line count, indentation, and word/line lengths are unchanged — so
1029
+ * the neutralized variant has the same layout and (for ASCII) the same byte
1030
+ * footprint as the input, but no readable content. Whitespace is preserved
1031
+ * deliberately: collapsing it would change the token structure and stop the
1032
+ * variant from being a true footprint match.
1033
+ */
1034
+ declare function neutralizeText(content: string): string;
1035
+ //#endregion
1036
+ //#region src/artifact-validator.d.ts
1037
+ /**
1038
+ * Artifact validators.
1039
+ *
1040
+ * Generic "score a produced artifact" primitive. Tax uses it for PDF form
1041
+ * correctness, research for sourced briefs, browser for task assertions, coding
1042
+ * for social posts. One interface, many validators; all plug into
1043
+ * `BenchmarkRunner` the same way.
1044
+ *
1045
+ * A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
1046
+ * plus a `ValidationContext` (scenario id, the turns that produced it) and
1047
+ * returns a `ValidationResult` with pass/fail + 0..1 score + structured
1048
+ * issues.
1049
+ */
1050
+ interface Artifact {
1051
+ /** Logical kind — validators type-guard on this */
1052
+ kind: 'file' | 'json' | 'text' | 'binary' | string;
1053
+ /** Filesystem-style path, optional */
1054
+ path?: string;
1055
+ /** String content for text/json/file kinds */
1056
+ content?: string;
1057
+ /** Binary content (if kind === 'binary') */
1058
+ bytes?: Uint8Array;
1059
+ /** Caller-supplied metadata (mimeType, sha256, size, etc.) */
1060
+ metadata?: Record<string, unknown>;
1061
+ }
1062
+ interface ValidationContext {
1063
+ scenarioId: string;
1064
+ turnIndex?: number;
1065
+ /** Prior artifacts for multi-artifact scenarios */
1066
+ priorArtifacts?: Artifact[];
1067
+ /** Free-form hints the validator uses for domain-specific checks */
1068
+ hints?: Record<string, unknown>;
1069
+ }
1070
+ interface ValidationIssue {
1071
+ severity: 'error' | 'warning' | 'info';
1072
+ message: string;
1073
+ /** Optional path into the artifact (e.g. JSON path or byte offset) */
1074
+ locus?: string;
1075
+ }
1076
+ interface ValidationResult {
1077
+ pass: boolean;
1078
+ /** 0–1 normalized score. Validators should be monotonic in pass-ness. */
1079
+ score: number;
1080
+ issues: ValidationIssue[];
1081
+ /** Diagnostic payload for reporters */
1082
+ evidence?: Record<string, unknown>;
1083
+ }
1084
+ interface ArtifactValidator {
1085
+ /** Stable identifier for the validator; appears in reports. */
1086
+ name: string;
1087
+ /** Optional description for human-facing reports. */
1088
+ description?: string;
1089
+ /** Called once per artifact; validators are expected to be pure + idempotent. */
1090
+ validate(artifact: Artifact, context: ValidationContext): Promise<ValidationResult>;
1091
+ }
1092
+ /**
1093
+ * Run every validator on the same artifact; aggregate pass as AND, score as
1094
+ * (weighted) mean, issues concatenated. Weights default to 1 each.
1095
+ */
1096
+ declare function composeValidators(validators: ArtifactValidator[], options?: {
1097
+ name?: string;
1098
+ weights?: number[];
1099
+ }): ArtifactValidator;
1100
+ /** Pass if the artifact body matches a provided regex. */
1101
+ declare function regexMatch(name: string, pattern: RegExp): ArtifactValidator;
1102
+ /** Pass if JSON parses and every required key is present. */
1103
+ declare function jsonHasKeys(name: string, requiredPaths: string[]): ArtifactValidator;
1104
+ /** Pass if min ≤ byte length ≤ max. */
1105
+ declare function byteLengthRange(name: string, min: number, max: number): ArtifactValidator;
1106
+ /** Pass if the artifact contains every required substring (case-insensitive by default). */
1107
+ declare function containsAll(name: string, required: string[], options?: {
1108
+ caseSensitive?: boolean;
1109
+ }): ArtifactValidator;
1110
+ //#endregion
1111
+ //#region src/completion-verifier.d.ts
1112
+ /** What kind of produced state can satisfy a requirement structurally. */
1113
+ type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
1114
+ interface CompletionRequirement {
1115
+ /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
1116
+ reqId: string;
1117
+ /** Human-readable description of the required deliverable. */
1118
+ title: string;
1119
+ /** Optional kind/category hint, matched against a produced item's kind. */
1120
+ category?: string;
1121
+ /** What produced state satisfies this requirement. Defaults to 'any'. */
1122
+ satisfiedBy?: SatisfiedBy;
1123
+ }
1124
+ interface TaskGold {
1125
+ taskId: string;
1126
+ requirements: CompletionRequirement[];
1127
+ }
1128
+ interface ProducedProposal {
1129
+ id: string;
1130
+ title: string;
1131
+ status: 'pending' | 'approved' | 'rejected';
1132
+ /** Optional persisted body — when present, enables a correctness check. */
1133
+ content?: string;
1134
+ }
1135
+ /** Everything observable about what a run actually produced. */
1136
+ interface ProducedState {
1137
+ /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
1138
+ artifacts: Artifact[];
1139
+ /** Proposals / filings the agent created. */
1140
+ proposals: ProducedProposal[];
1141
+ /** Names of tools the agent invoked. */
1142
+ toolCalls: string[];
1143
+ }
1144
+ interface RequirementCheck {
1145
+ reqId: string;
1146
+ title: string;
1147
+ /** A produced item of the right kind matched the requirement, non-empty. */
1148
+ structurallyPresent: boolean;
1149
+ /**
1150
+ * Whether the matched item actually fulfils the requirement. `null` when
1151
+ * not structurally present, when the matched item carries no content
1152
+ * to assess, or when the correctness check itself failed (`unmeasured`).
1153
+ */
1154
+ correct: boolean | null;
1155
+ /** structurallyPresent && !unmeasured && correct !== false. */
1156
+ satisfied: boolean;
1157
+ /**
1158
+ * Set when the correctness check itself errored (LLM call failure or an
1159
+ * unparseable response after retry). The requirement's fulfilment is
1160
+ * UNKNOWN — `correct` stays null, `satisfied` is false, and
1161
+ * `completionVerdict` excludes the row from `completionRate`'s
1162
+ * denominator. Never folded into a zero: a synthetic zero is
1163
+ * indistinguishable from a real failure (see `JudgeParseError`).
1164
+ */
1165
+ unmeasured?: true;
1166
+ /** Why the correctness check could not be measured (present iff `unmeasured`). */
1167
+ unmeasuredReason?: string;
1168
+ /** Human-readable evidence for the verdict. */
1169
+ evidence: string[];
1170
+ }
1171
+ /** Extends the substrate verdict spine: `valid` = `fullyComplete` and
1172
+ * `score` = `completionRate` — derived in `completionVerdict()`, the one
1173
+ * place those equalities hold by construction. */
1174
+ interface CompletionVerdict extends DefaultVerdict {
1175
+ taskId: string;
1176
+ requirements: RequirementCheck[];
1177
+ /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
1178
+ completionRate: number;
1179
+ /** Every measurable requirement satisfied (false when anything is unmeasured). */
1180
+ fullyComplete: boolean;
1181
+ /** Requirements whose correctness check errored — reported, never scored as zero. */
1182
+ unmeasuredCount: number;
1183
+ }
1184
+ /**
1185
+ * Construct a `CompletionVerdict` from the per-requirement checks, deriving
1186
+ * `completionRate` / `fullyComplete` and the spine fields (`valid` =
1187
+ * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
1188
+ * requirements — a verdict over nothing is a misconfiguration, mirroring
1189
+ * `verifyCompletion`'s gold-spec guard.
1190
+ */
1191
+ declare function completionVerdict(input: {
1192
+ taskId: string;
1193
+ requirements: RequirementCheck[];
1194
+ }): CompletionVerdict;
1195
+ /**
1196
+ * Decides whether a produced item's content actually fulfils a requirement.
1197
+ * Injected so the structural verifier stays pure and unit-testable; the
1198
+ * production implementation is `createLlmCorrectnessChecker`.
1199
+ */
1200
+ type CorrectnessChecker = (requirement: CompletionRequirement, content: string) => Promise<{
1201
+ correct: boolean;
1202
+ reason: string;
1203
+ }>;
1204
+ /**
1205
+ * Verify whether a run completed the task. `checkCorrectness` is injected —
1206
+ * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
1207
+ *
1208
+ * Throws on a gold spec with no requirements: an eval task that requires
1209
+ * nothing is a misconfiguration, not a vacuously-complete task.
1210
+ */
1211
+ declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
1212
+ interface LlmCorrectnessCheckerOpts {
1213
+ model?: string;
1214
+ /** Optional ledger for direct use. */
1215
+ costLedger?: CostLedgerHandle;
1216
+ costPhase?: string;
1217
+ costTags?: Record<string, string>;
1218
+ signal?: AbortSignal;
1219
+ /** Max chars of artifact content sent to the checker. */
1220
+ maxContentChars?: number;
1221
+ /**
1222
+ * Checker LLM calls per requirement before giving up (parse failures and
1223
+ * call errors both consume attempts). The failure then surfaces as an
1224
+ * `unmeasured` requirement, never a zero.
1225
+ */
1226
+ maxAttempts?: number;
1227
+ /**
1228
+ * Forensic capture of every checker request/response/error — without it a
1229
+ * checker failure is unauditable (the agent-turn raws never contain the
1230
+ * checker's own calls). Same sink contract as `LlmClient`.
1231
+ */
1232
+ rawSink?: RawProviderSink;
1233
+ }
1234
+ /**
1235
+ * Parse the correctness checker's model response. Tolerates a response
1236
+ * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
1237
+ * verdict boolean usually lands in the first few tokens, so a recovered
1238
+ * prefix with a boolean `correct` is a real measurement, not a guess.
1239
+ * Fails loud (JudgeParseError) when no boolean verdict is recoverable.
1240
+ */
1241
+ declare function parseCorrectnessResponse(raw: string): {
1242
+ correct: boolean;
1243
+ reason: string;
1244
+ };
1245
+ /**
1246
+ * Production `CorrectnessChecker` — one LLM call per matched artifact,
1247
+ * deterministic (temperature 0), structured JSON out. Judges fulfilment
1248
+ * only: a plan, a gesture, or a description of what should be done does not
1249
+ * fulfil a requirement — the artifact must BE the deliverable.
1250
+ */
1251
+ declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
1252
+ /**
1253
+ * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
1254
+ * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
1255
+ * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
1256
+ * of the requirement title's significant tokens. No network.
1257
+ *
1258
+ * Polarity-blind: token recall credits a negation that contains the
1259
+ * requirement's tokens ("I will NOT produce the comparison" recalls every token
1260
+ * of "produce the comparison"). The structural match stage is ALSO lexical, so
1261
+ * pairing the two collapses to a single gameable gate. Use this only as an
1262
+ * opt-in structural pre-filter or for tasks whose requirements have no polarity
1263
+ * to invert; for produced-state grading the correctness checker MUST be semantic
1264
+ * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
1265
+ */
1266
+ declare function createTokenRecallChecker(opts?: {
1267
+ minRecall?: number;
1268
+ minContentLength?: number;
1269
+ }): CorrectnessChecker;
1270
+ //#endregion
1271
+ //#region src/produced-state.d.ts
1272
+ /** A tool the agent invoked. */
1273
+ interface ToolCallEventLike {
1274
+ type: 'tool_call';
1275
+ toolName: string;
1276
+ }
1277
+ /**
1278
+ * An artifact the agent produced. `content` is the enriched field — the
1279
+ * runtime's base `artifact` event carries only metadata; the completion
1280
+ * oracle needs the body to verify the deliverable, so the runtime emits it.
1281
+ */
1282
+ interface ArtifactEventLike {
1283
+ type: 'artifact';
1284
+ artifactId: string;
1285
+ name?: string;
1286
+ mimeType?: string;
1287
+ uri?: string;
1288
+ content?: string;
1289
+ }
1290
+ /** A proposal / filing the agent created. */
1291
+ interface ProposalEventLike {
1292
+ type: 'proposal_created';
1293
+ proposalId: string;
1294
+ title: string;
1295
+ status?: 'pending' | 'approved' | 'rejected';
1296
+ content?: string;
1297
+ }
1298
+ /**
1299
+ * The subset of runtime stream events `extractProducedState` consumes.
1300
+ * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
1301
+ * the `{ type: string }` catch-all keeps the input permissive so callers can
1302
+ * pass the whole unfiltered telemetry stream — unrecognized events are skipped.
1303
+ */
1304
+ type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
1305
+ type: string;
1306
+ };
1307
+ /**
1308
+ * Normalize a run's runtime event stream into `ProducedState`.
1309
+ *
1310
+ * Pure and total — unrecognized event types are skipped. `toolCalls` is
1311
+ * deduplicated by name in first-seen order (completion cares about a tool's
1312
+ * presence, not its call count). An artifact with neither a name nor a uri
1313
+ * still yields an entry keyed by its `artifactId` so it is never silently
1314
+ * dropped; an artifact with no `content` yields empty content, which the
1315
+ * completion oracle's structural check then rejects on its own.
1316
+ */
1317
+ declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
1318
+ //#endregion
1319
+ //#region src/agent-profile.d.ts
1320
+ /**
1321
+ * The agentic coding harnesses an eval sweeps by default — the ones we care about
1322
+ * ranking. This is the SINGLE source of that list; consumers import it instead of
1323
+ * re-declaring their own (a re-declared list is how the fleet drifts). Pass an
1324
+ * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
1325
+ * harness) to widen beyond these.
1326
+ */
1327
+ declare const CODING_HARNESSES: readonly HarnessType[];
1328
+ interface ProfileAxisSpec {
1329
+ /** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
1330
+ * harness and model vary. `model.default` is the fallback model. */
1331
+ base: AgentProfile;
1332
+ /** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
1333
+ harnesses?: readonly HarnessType[];
1334
+ /** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
1335
+ * single-model behaviour, so omitting this never changes an existing run. */
1336
+ models?: readonly string[];
1337
+ /** Force every (harness, model) pair verbatim, even ones the harness can't run —
1338
+ * for deliberately testing failure modes. Default (false): SNAP instead — a
1339
+ * vendor-locked harness runs only the swept models in its family, or its native
1340
+ * default when it supports none, so no harness is dropped and none gets a
1341
+ * guaranteed-failing foreign-model cell. */
1342
+ keepIncompatible?: boolean;
1343
+ }
1344
+ /** Model sentinel for a vendor-locked harness that supports none of the swept models:
1345
+ * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
1346
+ * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
1347
+ * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
1348
+ * table that would rot as router catalogs change. */
1349
+ declare const HARNESS_NATIVE_MODEL = "default";
1350
+ /**
1351
+ * Expand a base profile across the harness × model matrix into the `AgentProfile[]`
1352
+ * that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
1353
+ * which models do we evaluate" lives, so no product hand-rolls its own harness list
1354
+ * or column→profile mapping (the pattern that let those copies drift and silently
1355
+ * break the harness pivot).
1356
+ *
1357
+ * Each cell clones `base`, sets `model.default`, and stamps `metadata.harness` +
1358
+ * `metadata.harnessModel` (both hash-bearing, so every cell gets a distinct
1359
+ * `agentProfileId` row and results join back by harness/model via {@link harnessAxisOf}
1360
+ * with no hand-recomputed key). A vendor-locked harness snaps to its family's swept
1361
+ * models — or its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none —
1362
+ * so every requested harness runs; `keepIncompatible` forces every pair verbatim.
1363
+ *
1364
+ * Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
1365
+ * everything we care about" switch, identical in shape whether one harness or all.
1366
+ */
1367
+ declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
1368
+ /**
1369
+ * Read the (harness, model) a matrix cell ran under, off a profile or a result row's
1370
+ * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
1371
+ * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
1372
+ * this instead of recomputing an id (recomputing the wrong key is what broke the pivot
1373
+ * in the hand-rolled copies).
1374
+ */
1375
+ declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
1376
+ harness: HarnessType;
1377
+ model: string;
1378
+ } | undefined;
1379
+ /**
1380
+ * Collision-resistant, path-safe, human-readable profile id for eval artifacts.
1381
+ * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
1382
+ * keys, and directory names where two profiles must not collapse onto one row.
1383
+ * The suffix is the first 64 bits of the behaviour hash, enough for ordinary
1384
+ * eval matrices while keeping filenames readable.
1385
+ */
1386
+ declare function agentProfileId(profile: AgentProfile): string;
1387
+ /**
1388
+ * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
1389
+ * model id because run records reject bare/missing model aliases.
1390
+ */
1391
+ declare function agentProfileModelId(profile: AgentProfile): string;
1392
+ /**
1393
+ * Deterministic behaviour identity for the canonical
1394
+ * `@tangle-network/agent-interface` AgentProfile.
1395
+ *
1396
+ * `name` and `description` are labels and do not affect the hash. Profile
1397
+ * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
1398
+ * and extensions do affect the hash. Resource array order is hash-bearing
1399
+ * because mount order can change agent behaviour. Undefined fields are treated
1400
+ * as absent; explicit `null` fields remain hash-bearing.
1401
+ */
1402
+ declare function agentProfileHash(profile: AgentProfile): string;
1403
+ //#endregion
1404
+ //#region src/integrity/backend-integrity.d.ts
1405
+ interface BackendIntegrityReport {
1406
+ /** Total records inspected. */
1407
+ totalRecords: number;
1408
+ /** Records with input=0 AND output=0 (a stub fingerprint). */
1409
+ stubRecords: number;
1410
+ /** Records with nonzero token usage (real LLM activity). */
1411
+ realRecords: number;
1412
+ /** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
1413
+ uncostedRecords: number;
1414
+ /** Sum of input tokens across all records. */
1415
+ totalInputTokens: number;
1416
+ /** Sum of output tokens across all records. */
1417
+ totalOutputTokens: number;
1418
+ /** Sum of costUsd across all records. */
1419
+ totalCostUsd: number;
1420
+ /** Worst-case integrity verdict. */
1421
+ verdict: 'real' | 'mixed' | 'stub';
1422
+ /** Human-readable diagnosis suitable for terminal output. */
1423
+ diagnosis: string;
1424
+ }
1425
+ /**
1426
+ * Error thrown when an integrity assertion fails. Caller can pattern-match
1427
+ * by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
1428
+ * errors.
1429
+ */
1430
+ declare class BackendIntegrityError extends AgentEvalError {
1431
+ readonly report: BackendIntegrityReport;
1432
+ constructor(message: string, report: BackendIntegrityReport);
1433
+ }
1434
+ /**
1435
+ * Inspect a batch of RunRecords and return an integrity report. Pure
1436
+ * function — no I/O, no logging. The caller decides what to do with the
1437
+ * verdict (print warning, throw, gate CI, etc.).
1438
+ */
1439
+ declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
1440
+ /** Inspect settled agent calls from the canonical cost ledger. */
1441
+ declare function summarizeAgentReceiptIntegrity(receipts: ReadonlyArray<CostReceipt>): BackendIntegrityReport;
1442
+ /**
1443
+ * Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
1444
+ * shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
1445
+ * to also reject mixed verdicts (recommended for CI gates).
1446
+ *
1447
+ * Real backends pass through silently.
1448
+ */
1449
+ declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
1450
+ allowMixed?: boolean;
1451
+ }): BackendIntegrityReport;
1452
+ /** Reject a cost ledger with no real agent call or a partial stub run. */
1453
+ declare function assertRealAgentReceipts(receipts: ReadonlyArray<CostReceipt>, opts?: {
1454
+ allowMixed?: boolean;
1455
+ }): BackendIntegrityReport;
1456
+ //#endregion
1457
+ //#region src/campaign/presets/run-profile-matrix.d.ts
1458
+ /** Thrown when the matrix is misconfigured (no profiles, a profile whose model
1459
+ * lacks a snapshot version, etc.). Distinct from `BackendIntegrityError`,
1460
+ * which signals a stub backend at run time. */
1461
+ declare class ProfileMatrixError extends AgentEvalError {
1462
+ constructor(message: string);
1463
+ }
1464
+ /** Dispatch for one cell: render `profile` against `scenario`, returning the
1465
+ * artifact the judges score. Run LLM work through `ctx.cost.runPaidCall` —
1466
+ * the integrity check depends on its receipt. */
1467
+ type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile$1, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1468
+ interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
1469
+ /** Axis 3 — the agent-under-test configurations. Each is one column. */
1470
+ profiles: AgentProfile$1[];
1471
+ /** Axis 1 — the persona/scenario corpus, run against every profile. */
1472
+ scenarios: TScenario[];
1473
+ /** Renders one (profile, scenario) cell. */
1474
+ dispatch: ProfileDispatchFn<TScenario, TArtifact>;
1475
+ /** The scoring axis. */
1476
+ judges?: JudgeConfig<TArtifact, TScenario>[];
1477
+ /** Where each profile's campaign writes artifacts/traces. One subdir per
1478
+ * profile. */
1479
+ runDir: string;
1480
+ /** Git SHA the harness ran from — stamped onto every RunRecord (mandatory
1481
+ * for paper-grade records). */
1482
+ commitSha: string;
1483
+ /** Logical experiment id shared across the whole matrix so the promotion
1484
+ * gate can pair profiles on matched scenarios. Default: a hash of the
1485
+ * profile + scenario ids. */
1486
+ experimentId?: string;
1487
+ /** Which split these runs belong to. Default `'search'`. */
1488
+ splitTag?: RunSplitTag;
1489
+ /** Replicates per (profile, scenario) cell for CI bands. Default 1. */
1490
+ reps?: number;
1491
+ /** Campaign seed (per profile). Default 42. */
1492
+ seed?: number;
1493
+ /**
1494
+ * Backend-integrity posture, enforced AFTER the matrix completes:
1495
+ * - `'assert'` (default) — throw `BackendIntegrityError` if the run was a
1496
+ * stub (and, with `allowMixed:false`, if it was mixed).
1497
+ * - `'warn'` — log the verdict but never throw.
1498
+ * - `'off'` — skip the guard entirely (only for offline/replay analysis).
1499
+ */
1500
+ integrity?: 'assert' | 'warn' | 'off';
1501
+ /** Forwarded to `assertRealBackend`. Default true (tolerate partial 429
1502
+ * cascades); set false for strict CI gates. */
1503
+ allowMixed?: boolean;
1504
+ /** Max concurrent cells WITHIN each profile's campaign. Default 2.
1505
+ * Profiles run sequentially so the cost ceiling is honored deterministically. */
1506
+ maxConcurrency?: number;
1507
+ /** Cumulative USD cap per profile campaign. */
1508
+ costCeiling?: number;
1509
+ /** Capture flywheel — forwarded to each campaign. */
1510
+ labeledStore?: LabeledScenarioStore | 'off';
1511
+ captureSource?: LabeledScenarioSource;
1512
+ /** Storage backend. Default `fsCampaignStorage`. Pass
1513
+ * `inMemoryCampaignStorage()` for edge/CF-Worker/test runs. */
1514
+ storage?: CampaignStorage;
1515
+ /** Test seam — override the wall clock. */
1516
+ now?: () => Date;
1517
+ /** Optional persona key per scenario — drives the `byPersona` pivot. When
1518
+ * unset, `byPersona` is omitted. */
1519
+ personaOf?: (scenario: TScenario) => string;
1520
+ /** Validate every produced RunRecord with `validateRunRecord` (fail-loud).
1521
+ * Default true — catches bad model snapshots and non-finite judge dims at
1522
+ * the boundary instead of letting them poison downstream analysis. */
1523
+ validate?: boolean;
1524
+ /** Corpus-by-default: derive the trajectory text (`prompt` + `completion`)
1525
+ * for each cell from its artifact + scenario. When set, every produced
1526
+ * record carries `prompt`/`completion` (a `CorpusRecord`) so the run's
1527
+ * graded trajectories can be appended to the durable RL corpus with no
1528
+ * side-channel — `appendToCorpus(result.records, path)`. Fail-soft: a
1529
+ * throwing or undefined-returning extractor just omits the text. */
1530
+ corpusText?: (artifact: TArtifact, scenario: TScenario) => {
1531
+ prompt: string;
1532
+ completion: string;
1533
+ } | undefined;
1534
+ }
1535
+ interface ProfileSummary {
1536
+ profileId: string;
1537
+ profileHash: string;
1538
+ model: string;
1539
+ /** RunRecords produced for this profile (= scenarios × reps). */
1540
+ records: number;
1541
+ /** Mean across scored records, or null when the profile has no task labels. */
1542
+ meanComposite: number | null;
1543
+ totalCostUsd: number;
1544
+ /** Per-profile integrity verdict — surfaces a single profile that ran stub
1545
+ * even when the matrix as a whole looks real. */
1546
+ integrity: BackendIntegrityReport;
1547
+ }
1548
+ interface ScenarioRollup {
1549
+ meanComposite: number;
1550
+ n: number;
1551
+ }
1552
+ interface RunProfileMatrixResult<TArtifact, TScenario extends Scenario> {
1553
+ matrixId: string;
1554
+ experimentId: string;
1555
+ /** One RunRecord per (profile, scenario, rep) cell — the integrity-checked,
1556
+ * paper-grade output. Feed straight into `analyzeRuns`, `HeldOutGate`,
1557
+ * scorecards, the hosted wire format. */
1558
+ records: RunRecord[];
1559
+ byProfile: Record<string, ProfileSummary>;
1560
+ byScenario: Record<string, ScenarioRollup>;
1561
+ /** Present only when `personaOf` was supplied. */
1562
+ byPersona?: Record<string, ScenarioRollup>;
1563
+ /** Whole-matrix integrity report (the one `integrity:'assert'` enforces). */
1564
+ integrity: BackendIntegrityReport;
1565
+ /** The raw per-profile campaign results, keyed by profile id. */
1566
+ campaigns: Record<string, CampaignResult<TArtifact, TScenario>>;
1567
+ }
1568
+ /**
1569
+ * Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
1570
+ */
1571
+ declare function runProfileMatrix<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixOptions<TScenario, TArtifact>): Promise<RunProfileMatrixResult<TArtifact, TScenario>>;
1572
+ //#endregion
1573
+ //#region src/campaign/presets/playback.d.ts
1574
+ /** One step of a user story — what the user does. The driver interprets
1575
+ * `payload` (a Playwright selector + action, or a sandbox chat turn). */
1576
+ interface PlaybackStep {
1577
+ /** Human-readable action, captured verbatim in the UX narrative. */
1578
+ action: string;
1579
+ /** Driver-specific payload (e.g. `{ selector, fill }` or `{ turn }`). */
1580
+ payload?: Record<string, unknown>;
1581
+ }
1582
+ /**
1583
+ * A user story = a runnable product journey plus the requirements that define
1584
+ * "this story works". Each requirement is one Jira ticket line. Extends
1585
+ * `Scenario` so a catalog drops straight into `runProfileMatrix({ scenarios })`.
1586
+ */
1587
+ interface UserStory extends Scenario {
1588
+ /** Human-readable story title (the ticket headline). */
1589
+ title: string;
1590
+ /** Ordered steps the driver executes. */
1591
+ steps: PlaybackStep[];
1592
+ /** What must hold in the produced state for the story to pass. */
1593
+ requirements: CompletionRequirement[];
1594
+ }
1595
+ /** Dispatch context plus the profile under test (which cheap model, etc.). */
1596
+ interface PlaybackContext extends DispatchContext {
1597
+ profile: AgentProfile;
1598
+ }
1599
+ /**
1600
+ * Drives the real product through a story and returns the runtime event stream
1601
+ * `extractProducedState` consumes. Implemented by CONSUMERS —
1602
+ * `SandboxPlaybackDriver` (real API / sandbox workspace) and
1603
+ * `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
1604
+ * browser infra the substrate must not import. The driver MUST report LLM
1605
+ * usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
1606
+ * tokens (a run that never reports tokens reads as a stub).
1607
+ */
1608
+ interface PlaybackDriver<TStory extends UserStory = UserStory> {
1609
+ run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
1610
+ }
1611
+ /**
1612
+ * Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
1613
+ * matrix scores is the `ProducedState` extracted from the driver's event
1614
+ * stream — grade it with `scoreUserStory` (or a judge wrapping it).
1615
+ */
1616
+ declare function makePlaybackDispatch<TStory extends UserStory>(driver: PlaybackDriver<TStory>): ProfileDispatchFn<TStory, ProducedState>;
1617
+ /** A scored user story — the completion verdict plus its human title. */
1618
+ interface UserStoryVerdict extends CompletionVerdict {
1619
+ title: string;
1620
+ }
1621
+ /**
1622
+ * Score one story's produced state against its requirements. Thin wrapper over
1623
+ * `verifyCompletion` that builds the gold from the story and returns a
1624
+ * per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
1625
+ * deterministic stub in tests, `createLlmCorrectnessChecker` in production.
1626
+ */
1627
+ declare function scoreUserStory(story: UserStory, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<UserStoryVerdict>;
1628
+ /** One row of the launch scoreboard — story × requirement → PASS/FAIL. */
1629
+ interface ScoreboardRow {
1630
+ storyId: string;
1631
+ storyTitle: string;
1632
+ reqId: string;
1633
+ reqTitle: string;
1634
+ status: 'PASS' | 'FAIL';
1635
+ evidence: string[];
1636
+ }
1637
+ /**
1638
+ * Flatten story verdicts into the per-requirement scoreboard — the literal
1639
+ * Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
1640
+ * evidence behind the verdict.
1641
+ */
1642
+ declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
1643
+ /** Launch-readiness headline counts rolled up from the per-requirement rows. */
1644
+ interface ScoreboardSummary {
1645
+ /** Distinct user stories on the board. */
1646
+ stories: number;
1647
+ /** Stories whose every requirement passed. */
1648
+ storiesFullyComplete: number;
1649
+ /** Total (story, requirement) rows. */
1650
+ requirements: number;
1651
+ /** Rows with status PASS. */
1652
+ passed: number;
1653
+ /** Rows with status FAIL. */
1654
+ failed: number;
1655
+ /** passed / requirements; 0 when there are no rows. */
1656
+ passRate: number;
1657
+ }
1658
+ /** Roll the per-requirement rows up into the launch headline counts. */
1659
+ declare function scoreboardSummary(rows: readonly ScoreboardRow[]): ScoreboardSummary;
1660
+ interface ScoreboardRenderOptions {
1661
+ /** Document H1. Defaults to a generic playback title. */
1662
+ title?: string;
1663
+ /** Key/value run metadata rendered under the headline (runId, backend, model, date). */
1664
+ meta?: Record<string, string>;
1665
+ /** Max chars of joined evidence shown per row. Default 160. */
1666
+ maxEvidenceChars?: number;
1667
+ }
1668
+ /**
1669
+ * Render the scoreboard as a launch-readiness Markdown document — the literal
1670
+ * "tick off every user story" artifact: a headline roll-up, the open tickets
1671
+ * (FAIL rows) up top as the launch blockers, then a per-story table of
1672
+ * requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
1673
+ * rows in, same bytes out (no clock/random), so it is safe to snapshot.
1674
+ */
1675
+ declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
1676
+ //#endregion
1677
+ //#region src/campaign/run-dir.d.ts
1678
+ /** The shared, out-of-repo root for campaign/benchmark run bundles. Keeping run
1679
+ * outputs here means they never land in a repo working tree (no per-repo
1680
+ * gitignore, no clutter, no accidental commits). Layout:
1681
+ * ~/.tangle/traces/<repo>/runs/<runName>/
1682
+ * where <repo> disambiguates runs across repos in one place. */
1683
+ declare function tangleTracesRoot(): string;
1684
+ /** Resolve a campaign `runDir`. An absolute path is honored as-is (the caller
1685
+ * chose an explicit location). A bare name is placed under the shared home root
1686
+ * so bundles never pollute a repo working tree — the default the harness should
1687
+ * compute so callers pass a *name*, not a path. */
1688
+ declare function resolveRunDir(runDir: string, repo?: string): string;
1689
+ //#endregion
1690
+ //#region src/campaign/scenario-selection.d.ts
1691
+ /**
1692
+ * Discriminative scenario selection (research claim E2).
1693
+ *
1694
+ * The OR benchmark is SATURATING: run 7 measured ~75% tied holdout cells — most
1695
+ * problems are solved optimally by the baseline AND every candidate, so those
1696
+ * paired cells carry zero signal. A random/balanced holdout split spends its
1697
+ * budget on scenarios that cannot separate candidates.
1698
+ *
1699
+ * This picks the holdout by DISCRIMINATION power instead: a scenario every
1700
+ * candidate scores identically (variance ~0) carries no signal; one where the
1701
+ * scores spread carries the most. We drop fully saturated ties so each paired
1702
+ * holdout cell is spent on a scenario that can actually move a verdict.
1703
+ */
1704
+ /** Per-scenario observation: the composite scores each candidate earned on it. */
1705
+ interface ScenarioSignal {
1706
+ scenarioId: string;
1707
+ /** Per-candidate composite scores observed for this scenario (>=1 values). */
1708
+ scores: number[];
1709
+ }
1710
+ interface DiscriminationScore {
1711
+ scenarioId: string;
1712
+ /** Higher = separates candidates more (spread of their scores). */
1713
+ discrimination: number;
1714
+ /** Higher = easier / more-saturated (mean of candidate scores). */
1715
+ meanScore: number;
1716
+ variance: number;
1717
+ /** variance ~0 AND meanScore at/above the ceiling ⇒ a saturated tie, no signal. */
1718
+ tied: boolean;
1719
+ }
1720
+ /**
1721
+ * Rank scenarios by how well they DISCRIMINATE candidates.
1722
+ *
1723
+ * `discrimination = variance` (spread of the candidate scores) — kept simple on
1724
+ * purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
1725
+ * so that, among equally spread scenarios, the one with more room to improve
1726
+ * ranks first. Returned sorted by the deterministic order above.
1727
+ */
1728
+ declare function scoreDiscrimination(signals: ScenarioSignal[], opts?: {
1729
+ saturationCeiling?: number;
1730
+ }): DiscriminationScore[];
1731
+ /**
1732
+ * Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
1733
+ * fully saturated ties when enough non-tied scenarios exist (a tie in the
1734
+ * holdout wastes a paired cell).
1735
+ *
1736
+ * Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
1737
+ * least-saturated tied ones (tied scenarios are already ordered least-saturated
1738
+ * first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
1739
+ * `signals.length <= k`, returns all ids in discrimination order.
1740
+ */
1741
+ declare function selectDiscriminative(signals: ScenarioSignal[], k: number, opts?: {
1742
+ saturationCeiling?: number;
1743
+ }): string[];
1744
+ //#endregion
1745
+ //#region src/campaign/score-utils.d.ts
1746
+ /** Mean composite across cells with complete task-quality evidence.
1747
+ * Partial judge results remain on their cells but never enter this value.
1748
+ * A campaign with no complete score has no numeric mean and fails loudly. */
1749
+ declare function campaignMeanComposite<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): number;
1750
+ /** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
1751
+ * Returns a positive number when `a` ranks above `b`, negative when below, and
1752
+ * zero when equal. */
1753
+ declare function compareRankKeys(a: readonly number[], b: readonly number[]): number;
1754
+ interface CampaignBreakdown {
1755
+ /** Mean score per judge dimension across all cells. */
1756
+ dimensions: Record<string, number>;
1757
+ /** Per-scenario composite (mean over reps + judges) + the judge's free-form
1758
+ * `notes` for that scenario (the "why" a reflective proposer grounds on) +
1759
+ * an optional `emitted` excerpt of the candidate's raw output (the "what it
1760
+ * actually did" a reflective proposer grounds on). */
1761
+ scenarios: Array<{
1762
+ scenarioId: string;
1763
+ composite: number;
1764
+ notes?: string;
1765
+ emitted?: string;
1766
+ }>;
1767
+ }
1768
+ /** Per-candidate evidence a reflective/patch proposer grounds its next proposal
1769
+ * on: mean score per judge dimension + per-scenario composite. */
1770
+ declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
1771
+ //#endregion
1772
+ //#region src/campaign/search-ledger-errors.d.ts
1773
+ declare class SearchLedgerError extends ValidationError {}
1774
+ declare class SearchLedgerIntegrityError extends SearchLedgerError {}
1775
+ declare class SearchLedgerConflictError extends SearchLedgerError {}
1776
+ //#endregion
1777
+ //#region src/campaign/search-ledger.d.ts
1778
+ declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
1779
+ type SearchLedgerHash = LedgerHash;
1780
+ type SearchSurfaceKind = 'prompt' | 'tool-contract' | 'runtime-config' | 'memory' | 'knowledge' | 'agent-profile' | 'code' | 'deployment';
1781
+ /** Content-addressed artifact or receipt. Mutable paths are locators only; the
1782
+ * digest and byte length bind the exact bytes used by the search. */
1783
+ interface SearchArtifactRef {
1784
+ role: string;
1785
+ uri: string;
1786
+ sha256: SearchLedgerHash;
1787
+ byteLength: number;
1788
+ }
1789
+ /** Repository, dataset, or package source pinned to an immutable commit or
1790
+ * content digest. Branches, tags, and bare package versions are rejected. */
1791
+ interface SearchSourceRef {
1792
+ uri: string;
1793
+ revision: string;
1794
+ }
1795
+ interface SearchModelIdentity {
1796
+ provider: string;
1797
+ snapshot: string;
1798
+ }
1799
+ interface SearchCandidateSurface {
1800
+ surfaceId: string;
1801
+ kind: SearchSurfaceKind;
1802
+ artifact: SearchArtifactRef;
1803
+ }
1804
+ interface SearchCandidateLineage {
1805
+ /** Existing `LineageNode.id`; this ledger references rather than embeds it. */
1806
+ lineageNodeId: string;
1807
+ parentCandidateIds: string[];
1808
+ generation: number;
1809
+ proposer: string;
1810
+ proposerSource: SearchSourceRef;
1811
+ }
1812
+ type SearchOperationKind = 'candidate-generation' | 'analysis' | 'selection' | 'judge' | 'other';
1813
+ interface SearchPlannedTask {
1814
+ taskId: string;
1815
+ source: SearchSourceRef;
1816
+ benchmark: SearchSourceRef;
1817
+ /** Maximum transport attempts for this task and candidate. Only an explicit
1818
+ * passed/failed outcome satisfies the planned denominator. */
1819
+ maxAttempts: number;
1820
+ }
1821
+ interface SearchPlannedOperation {
1822
+ operationId: string;
1823
+ kind: SearchOperationKind;
1824
+ }
1825
+ interface SearchCandidateSlot {
1826
+ slotId: string;
1827
+ /** Planned candidate-generation call that must either produce this slot or
1828
+ * fail before the slot can be closed. Several slots may share one batched call. */
1829
+ generationOperationId: string;
1830
+ }
1831
+ interface SearchPlan {
1832
+ /** Stable slots and their proposer calls are frozen before search begins. */
1833
+ candidateSlots: SearchCandidateSlot[];
1834
+ /** Every task applies to every successfully registered candidate. */
1835
+ tasks: SearchPlannedTask[];
1836
+ /** Non-task spend slots: proposal, analysis, selection, extra judges, etc. */
1837
+ operations: SearchPlannedOperation[];
1838
+ }
1839
+ type SearchTokenAccounting = {
1840
+ status: 'known';
1841
+ inputTokens: number;
1842
+ outputTokens: number;
1843
+ cachedTokens: number;
1844
+ } | {
1845
+ status: 'unknown';
1846
+ reason: string;
1847
+ };
1848
+ type SearchCostAccounting = {
1849
+ status: 'known';
1850
+ usd: number;
1851
+ source: 'provider' | 'pricing-table' | 'free';
1852
+ } | {
1853
+ status: 'unknown';
1854
+ /** Known spend may still be a lower bound when one call was unpriced. */
1855
+ knownLowerBoundUsd: number;
1856
+ reason: string;
1857
+ };
1858
+ interface SearchAttemptAccounting {
1859
+ tokens: SearchTokenAccounting;
1860
+ cost: SearchCostAccounting;
1861
+ }
1862
+ interface SearchFailureReason {
1863
+ code: string;
1864
+ message: string;
1865
+ }
1866
+ type SearchTaskOutcome = {
1867
+ status: 'passed';
1868
+ score: number;
1869
+ metrics: Record<string, number>;
1870
+ } | {
1871
+ status: 'failed';
1872
+ score: number;
1873
+ metrics: Record<string, number>;
1874
+ failure: SearchFailureReason;
1875
+ } | {
1876
+ status: 'errored';
1877
+ metrics: Record<string, number>;
1878
+ error: SearchFailureReason & {
1879
+ retryable: boolean;
1880
+ };
1881
+ };
1882
+ type SearchSurfaceEffect = {
1883
+ status: 'measured';
1884
+ metric: string;
1885
+ baselineValue: number;
1886
+ candidateValue: number;
1887
+ delta: number;
1888
+ } | {
1889
+ status: 'not-measured';
1890
+ reason: string;
1891
+ };
1892
+ /** Per-attempt proof that a declared candidate surface was or was not active,
1893
+ * plus measured effect when the experiment supports attribution. */
1894
+ interface SearchSurfaceEvidence {
1895
+ surfaceId: string;
1896
+ fired: boolean;
1897
+ firingCount: number;
1898
+ effect: SearchSurfaceEffect;
1899
+ evidence: SearchArtifactRef[];
1900
+ }
1901
+ interface SearchLedgerEventBase {
1902
+ eventId: string;
1903
+ occurredAt: string;
1904
+ artifacts: SearchArtifactRef[];
1905
+ }
1906
+ interface SearchPlannedEvent extends SearchLedgerEventBase {
1907
+ kind: 'search-planned';
1908
+ plan: SearchPlan;
1909
+ }
1910
+ interface SearchCandidateRegisteredEvent extends SearchLedgerEventBase {
1911
+ kind: 'candidate-registered';
1912
+ slotId: string;
1913
+ generationOperationId: string;
1914
+ candidateId: string;
1915
+ lineage: SearchCandidateLineage;
1916
+ surfaces: SearchCandidateSurface[];
1917
+ }
1918
+ interface SearchCandidateSlotClosedEvent extends SearchLedgerEventBase {
1919
+ kind: 'candidate-slot-closed';
1920
+ slotId: string;
1921
+ generationOperationId: string;
1922
+ reason: SearchFailureReason;
1923
+ }
1924
+ interface SearchTaskAttemptedEvent extends SearchLedgerEventBase {
1925
+ kind: 'task-attempted';
1926
+ candidateId: string;
1927
+ runId: string;
1928
+ attemptIndex: number;
1929
+ task: {
1930
+ taskId: string;
1931
+ source: SearchSourceRef;
1932
+ };
1933
+ identity: {
1934
+ model: SearchModelIdentity;
1935
+ agent: SearchSourceRef;
1936
+ benchmark: SearchSourceRef;
1937
+ };
1938
+ outcome: SearchTaskOutcome;
1939
+ accounting: SearchAttemptAccounting;
1940
+ surfaceEvidence: SearchSurfaceEvidence[];
1941
+ }
1942
+ interface SearchOperationRecordedEvent extends SearchLedgerEventBase {
1943
+ kind: 'search-operation-recorded';
1944
+ operationId: string;
1945
+ operationKind: SearchOperationKind;
1946
+ execution: {
1947
+ kind: 'model';
1948
+ model: SearchModelIdentity;
1949
+ source: SearchSourceRef;
1950
+ } | {
1951
+ kind: 'deterministic';
1952
+ source: SearchSourceRef;
1953
+ };
1954
+ outcome: {
1955
+ status: 'completed';
1956
+ } | {
1957
+ status: 'partial';
1958
+ failure: SearchFailureReason;
1959
+ } | {
1960
+ status: 'failed';
1961
+ failure: SearchFailureReason;
1962
+ };
1963
+ accounting: SearchAttemptAccounting;
1964
+ }
1965
+ interface SearchCandidateDecidedEvent extends SearchLedgerEventBase {
1966
+ kind: 'candidate-decided';
1967
+ candidateId: string;
1968
+ decision: {
1969
+ status: 'selected';
1970
+ } | {
1971
+ status: 'rejected';
1972
+ reason: SearchFailureReason;
1973
+ };
1974
+ }
1975
+ interface SearchCompletedEvent extends SearchLedgerEventBase {
1976
+ kind: 'search-completed';
1977
+ result: {
1978
+ status: 'selected';
1979
+ candidateId: string;
1980
+ } | {
1981
+ status: 'all-rejected';
1982
+ reason: SearchFailureReason;
1983
+ };
1984
+ }
1985
+ type SearchLedgerEvent = SearchPlannedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
1986
+ interface SearchLedgerEntry {
1987
+ schema: typeof SEARCH_LEDGER_SCHEMA;
1988
+ campaignId: string;
1989
+ sequence: number;
1990
+ previousHash: SearchLedgerHash | null;
1991
+ event: SearchLedgerEvent;
1992
+ entryHash: SearchLedgerHash;
1993
+ }
1994
+ type SearchAccountingAudit = {
1995
+ status: 'known';
1996
+ inputTokens: number;
1997
+ outputTokens: number;
1998
+ cachedTokens: number;
1999
+ costUsd: number;
2000
+ } | {
2001
+ status: 'partial';
2002
+ knownInputTokens: number;
2003
+ knownOutputTokens: number;
2004
+ knownCachedTokens: number;
2005
+ knownCostUsd: number;
2006
+ unknownTokenEventIds: string[];
2007
+ unknownCostEventIds: string[];
2008
+ };
2009
+ interface SearchLedgerAudit {
2010
+ campaignId: string;
2011
+ eventCount: number;
2012
+ candidateCount: number;
2013
+ closedCandidateSlotCount: number;
2014
+ attemptCount: number;
2015
+ operationCount: number;
2016
+ outcomes: {
2017
+ passed: number;
2018
+ failed: number;
2019
+ errored: number;
2020
+ };
2021
+ operationOutcomes: {
2022
+ completed: number;
2023
+ partial: number;
2024
+ failed: number;
2025
+ };
2026
+ decisions: {
2027
+ selected: number;
2028
+ rejected: number;
2029
+ pending: number;
2030
+ };
2031
+ expected: {
2032
+ candidateSlots: number;
2033
+ taskOutcomes: number;
2034
+ operations: number;
2035
+ missingCandidateSlots: string[];
2036
+ missingTaskOutcomes: string[];
2037
+ missingOperations: string[];
2038
+ };
2039
+ status: 'in-progress' | 'selected' | 'all-rejected';
2040
+ selectedCandidateId: string | null;
2041
+ accounting: SearchAccountingAudit;
2042
+ headHash: SearchLedgerHash | null;
2043
+ }
2044
+ interface SearchLedgerReplay {
2045
+ entries: SearchLedgerEntry[];
2046
+ plan: SearchPlannedEvent | null;
2047
+ candidates: SearchCandidateRegisteredEvent[];
2048
+ closedCandidateSlots: SearchCandidateSlotClosedEvent[];
2049
+ attempts: SearchTaskAttemptedEvent[];
2050
+ operations: SearchOperationRecordedEvent[];
2051
+ decisions: SearchCandidateDecidedEvent[];
2052
+ completion: SearchCompletedEvent | null;
2053
+ audit: SearchLedgerAudit;
2054
+ }
2055
+ interface SearchLedgerAppendResult {
2056
+ entry: SearchLedgerEntry;
2057
+ /** False when the exact event was already durably present. */
2058
+ appended: boolean;
2059
+ replay: SearchLedgerReplay;
2060
+ }
2061
+ /** Validate and return a canonical copy. Arrays whose order is not semantic are
2062
+ * sorted so retries from different processes produce byte-identical events. */
2063
+ declare function validateSearchLedgerEvent(input: unknown): SearchLedgerEvent;
2064
+ interface OpenSearchLedgerOptions {
2065
+ path: string;
2066
+ campaignId: string;
2067
+ }
2068
+ interface SearchLedger {
2069
+ readonly path: string;
2070
+ readonly campaignId: string;
2071
+ append(event: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
2072
+ replay(): Promise<SearchLedgerReplay>;
2073
+ }
2074
+ /** Open a durable filesystem search ledger. Construction performs no I/O; the
2075
+ * first `append` or `replay` validates the complete existing file. */
2076
+ declare function openSearchLedger(options: OpenSearchLedgerOptions): SearchLedger;
2077
+ declare class FileSearchLedger implements SearchLedger {
2078
+ readonly path: string;
2079
+ readonly campaignId: string;
2080
+ private readonly journal;
2081
+ constructor(path: string, campaignId: string);
2082
+ replay(): Promise<SearchLedgerReplay>;
2083
+ append(input: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
2084
+ }
2085
+ //#endregion
2086
+ //#region src/campaign/single-run-lock.d.ts
2087
+ /**
2088
+ * Single-run lock for evaluations that share one mutable environment.
2089
+ *
2090
+ * Two concurrent runs against a shared stateful gym silently corrupt each
2091
+ * other: each resets/mutates environment state mid-cell of the other, and
2092
+ * every score from both becomes garbage that LOOKS like worker variance
2093
+ * (agent-lab R357 burned hours on flip-flopping scores before tracing them
2094
+ * to exactly this). The fix is a pid lockfile: refuse to start while a live
2095
+ * holder exists, reclaim stale locks whose pid is gone, release only if the
2096
+ * lock is still ours.
2097
+ *
2098
+ * `alsoCheck` exists because independent runners can guard the same shared
2099
+ * resource with differently named lockfiles; a runner must respect all of
2100
+ * them even though it writes only its own.
2101
+ */
2102
+ interface SingleRunLockOptions {
2103
+ /** Lockfile this runner writes (and checks). */
2104
+ readonly lockPath: string;
2105
+ /** Other runners' lockfiles guarding the same resource; checked, never written. */
2106
+ readonly alsoCheck?: readonly string[];
2107
+ /** Install a process 'exit' hook that releases the lock. Default true. */
2108
+ readonly releaseOnExit?: boolean;
2109
+ /** Owner pid recorded in the lockfile metadata. Default process.pid. */
2110
+ readonly pid?: number;
2111
+ }
2112
+ interface SingleRunLock {
2113
+ /** Remove the lockfile if this process still owns it. Idempotent. */
2114
+ release(): void;
2115
+ }
2116
+ /**
2117
+ * Acquire the lock or throw naming the live holder. A stale lock (holder pid
2118
+ * no longer running) is reclaimed by one contender. An interrupted reclaim
2119
+ * leaves a marker that fails closed instead of admitting overlapping runs.
2120
+ */
2121
+ declare function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock;
2122
+ //#endregion
2123
+ //#region src/campaign/surface-identity.d.ts
2124
+ /** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
2125
+ declare function assertCodeSurfaceIdentity(surface: unknown): asserts surface is CodeSurface;
2126
+ declare function assertComponentSurface(surface: unknown): asserts surface is ComponentSurface;
2127
+ declare function componentSurfaceIdentityMaterial(surface: ComponentSurface): string;
2128
+ /** Canonical, location-independent identity of a finalized code candidate.
2129
+ * Commit metadata is excluded: two commits with the same base, final tree,
2130
+ * and patch bytes are the same executable candidate. */
2131
+ declare function codeSurfaceIdentityMaterial(surface: CodeSurface): string;
2132
+ /** Full SHA-256 content identity for a prompt or finalized code surface. */
2133
+ declare function surfaceContentHash(surface: MutableSurface): `sha256:${string}`;
2134
+ /** Short loop key derived from the same content identity as provenance. */
2135
+ declare function surfaceHash(surface: MutableSurface): string;
2136
+ /** Canonical customer-visible description of the exact before/after surfaces. */
2137
+ declare function renderSurfaceDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
2138
+ //#endregion
2139
+ //#region src/campaign/transient-failure.d.ts
2140
+ /**
2141
+ * Transient-transport-failure classification for dispatch retry policies.
2142
+ *
2143
+ * When an eval cell dies, the harness must decide: retry (the infrastructure
2144
+ * hiccuped - a 502 storm, an admission-queue rejection, a dropped stream) or
2145
+ * score it (the agent genuinely failed). Getting this wrong corrupts results
2146
+ * in both directions: scoring transport hiccups as failures buries real
2147
+ * effects under noise (agent-lab R353 found 5/30 identical repeats were 502s
2148
+ * scored as task failures), while retrying genuine failures silently drops
2149
+ * the hard cells and inflates every arm.
2150
+ *
2151
+ * Full-duration timeouts are the deliberate knob: on saturated shared
2152
+ * infrastructure a timeout usually means the request never got a slot
2153
+ * (retry it), but on unthrottled infrastructure it means the agent flailed
2154
+ * on the task until the clock ran out (a real score-0). Both readings were
2155
+ * needed in practice within one week, so the classifier takes it as an
2156
+ * option instead of hardcoding either.
2157
+ */
2158
+ interface TransientFailureOptions {
2159
+ /**
2160
+ * Treat full-duration timeouts ("timeout after 180000ms") as transient.
2161
+ * Enable on saturated shared infrastructure where queue starvation eats
2162
+ * the clock; leave off when the agent had the resources and simply failed.
2163
+ * Default false.
2164
+ */
2165
+ readonly retryFullDurationTimeouts?: boolean;
2166
+ /** Additional caller-specific transient patterns. */
2167
+ readonly extraPatterns?: readonly RegExp[];
2168
+ }
2169
+ /**
2170
+ * True when the error text describes an infrastructure hiccup that should be
2171
+ * retried rather than scored. Empty/undefined input is not transient.
2172
+ */
2173
+ declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
2174
+ //#endregion
2175
+ //#region src/campaign/worktree/index.d.ts
2176
+ type GitOutput = string | Uint8Array;
2177
+ type GitEnvironment = Readonly<Record<string, string>>;
2178
+ type GitRunner = (args: string[], cwd: string, env?: GitEnvironment) => GitOutput;
2179
+ interface Worktree {
2180
+ /** Absolute path to the checked-out worktree directory. */
2181
+ readonly path: string;
2182
+ /** The branch the worktree is on (becomes the PR branch on promotion). */
2183
+ readonly branch: string;
2184
+ /** The ref the worktree was forked from. */
2185
+ readonly baseRef: string;
2186
+ /** Exact commit `baseRef` resolved to before the worktree was created. */
2187
+ readonly baseCommit: string;
2188
+ /** Exact tree object for `baseCommit`. */
2189
+ readonly baseTree: string;
2190
+ }
2191
+ interface WorktreeAdapter {
2192
+ /** Create an isolated worktree on a fresh branch off `baseRef`. */
2193
+ create(opts: {
2194
+ baseRef: string;
2195
+ label: string;
2196
+ }): Promise<Worktree>;
2197
+ /** Commit pending changes, freeze the exact Git objects + binary patch, and
2198
+ * verify the worktree still matches that identity. */
2199
+ finalize(worktree: Worktree, summary: string): Promise<CodeSurface>;
2200
+ /** Idempotently remove the worktree and branch. Safe to retry after partial cleanup. */
2201
+ discard(worktree: Worktree): Promise<void>;
2202
+ }
2203
+ /** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
2204
+ declare class WorktreeAdapterError extends Error {
2205
+ readonly cause?: unknown;
2206
+ constructor(message: string, cause?: unknown);
2207
+ }
2208
+ interface GitWorktreeAdapterOptions {
2209
+ /** Repo root the worktrees fork from. */
2210
+ repoRoot: string;
2211
+ /** Directory worktrees are created under. Default: `<repoRoot>/.worktrees`. */
2212
+ worktreeDir?: string;
2213
+ /** Branch-name prefix. Default: `improve`. */
2214
+ branchPrefix?: string;
2215
+ /** Test seam — defaults to a real `git` runner. The return value must contain
2216
+ * stdout verbatim, and runners that execute Git must forward the optional
2217
+ * environment overrides used to isolate patch generation. */
2218
+ git?: GitRunner;
2219
+ }
2220
+ interface CodeSurfaceVerification {
2221
+ /** Verified worktree path. */
2222
+ path: string;
2223
+ /** Git's canonical root for the verified checkout. */
2224
+ repoRoot: string;
2225
+ /** Recomputed full content identity. */
2226
+ contentHash: `sha256:${string}`;
2227
+ /** Exact verified binary-patch bytes. Candidate-bundle builders encode this
2228
+ * directly instead of reproducing Git diff options. */
2229
+ patchBytes: Uint8Array;
2230
+ }
2231
+ /**
2232
+ * Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
2233
+ */
2234
+ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAdapter;
2235
+ /** Verify a finalized code surface against its current checkout. This rejects
2236
+ * dirty/ignored files, moved refs, missing Git objects, raw byte/mode
2237
+ * mismatches, external symlinks, and submodules. */
2238
+ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string): CodeSurfaceVerification;
2239
+ /** Resolve a code candidate for evaluation only after verifying its immutable
2240
+ * identity against the checkout at `worktreeRef`. */
2241
+ declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
2242
+ //#endregion
2243
+ export { SearchTaskAttemptedEvent as $, sequentialDecide as $n, CrossSurfaceRankedSingle as $r, ArtifactEventLike as $t, SearchCandidateSlot as A, FsLabeledScenarioStoreOptions as An, CrossSurfaceBootstrapPolicy as Ar, ProfileDispatchFn as At, SearchLedgerHash as B, HeldoutSignificance as Bn, CrossSurfaceEligibility as Br, assertRealBackend as Bt, SEARCH_LEDGER_SCHEMA as C, byteLengthRange as Cn, planEvalFixtureRun as Cr, UserStory as Ct, SearchCandidateDecidedEvent as D, regexMatch as Dn, CrossSurfaceAdditionRejectionReason as Dr, scoreUserStory as Dt, SearchAttemptAccounting as E, jsonHasKeys as En, CrossSurfaceAdditionDecision as Er, renderScoreboardMarkdown as Et, SearchFailureReason as F, ScoredRollout as Fn, CrossSurfaceCandidateSummary as Fr, ScenarioRollup as Ft, SearchPlan as G, heldoutSignificance as Gn, CrossSurfaceInteractionPath as Gr, HARNESS_NATIVE_MODEL as Gt, SearchModelIdentity as H, PairedHoldout as Hn, CrossSurfaceIneligibilityReason as Hr, summarizeBackendIntegrity as Ht, SearchLedger as I, UngroundedLiteralReport as In, CrossSurfaceComponent as Ir, runProfileMatrix as It, SearchPlannedTask as J, SequentialDecideOptions as Jn, CrossSurfaceNaiveStackSelection as Jr, agentProfileHash as Jt, SearchPlannedEvent as K, pairHoldout as Kn, CrossSurfaceInteractionReport as Kr, HarnessType$1 as Kt, SearchLedgerAppendResult as L, classifyUngroundedLiterals as Ln, CrossSurfaceComponentEvidence as Lr, BackendIntegrityError as Lt, SearchCandidateSurface as M, RolloutArgumentDiff as Mn, CrossSurfaceCandidateComparison as Mr, ProfileSummary as Mt, SearchCompletedEvent as N, RolloutArgumentDiffOptions as Nn, CrossSurfaceCandidateEvidence as Nr, RunProfileMatrixOptions as Nt, SearchCandidateLineage as O, neutralizeText as On, CrossSurfaceAttemptCompleteness as Or, scoreboardSummary as Ot, SearchCostAccounting as P, RolloutCall as Pn, CrossSurfaceCandidateOutcome as Pr, RunProfileMatrixResult as Pt, SearchSurfaceKind as Q, SequentialPairedGateOptions as Qn, CrossSurfacePairwiseEntry as Qr, harnessAxisOf as Qt, SearchLedgerEntry as R, rolloutArgumentDiff as Rn, CrossSurfaceCompositionStep as Rr, BackendIntegrityReport as Rt, OpenSearchLedgerOptions as S, failureModeRecallJudge as Si, ValidationResult as Sn, loadEvalFixtureScenarios as Sr, ScoreboardSummary as St, SearchArtifactRef as T, containsAll as Tn, AnalyzeCrossSurfaceInteractionsInput as Tr, makePlaybackDispatch as Tt, SearchOperationKind as U, detectScale as Un, CrossSurfaceInteractionAwareSelection as Ur, AgentProfile$1 as Ut, SearchLedgerReplay as V, HeldoutSignificanceOptions as Vn, CrossSurfaceEvidenceBreakdown as Vr, summarizeAgentReceiptIntegrity as Vt, SearchOperationRecordedEvent as W, dimensionRegressions as Wn, CrossSurfaceInteractionEffect as Wr, CODING_HARNESSES as Wt, SearchSurfaceEffect as X, SequentialObservation as Xn, CrossSurfacePairEvidence as Xr, agentProfileModelId as Xt, SearchSourceRef as Y, SequentialDecision as Yn, CrossSurfacePairCompatibility as Yr, agentProfileId as Yt, SearchSurfaceEvidence as Z, SequentialPairedGate as Zn, CrossSurfacePairIncompatibilityReason as Zr, expandProfileAxes as Zt, surfaceHash as _, AnalystArtifact as _i, verifyCompletion as _n, EvalFixtureValidationMode as _r, PlaybackContext as _t, WorktreeAdapterError as a, MatchedPair as ai, CompletionVerdict as an, canonicalize as ar, SearchLedgerError as at, acquireSingleRunLock as b, FailureModeRecallJudgeOptions as bi, ValidationContext as bn, discoverEvalFixtures as br, ScoreboardRenderOptions as bt, verifyCodeSurface as c, PairArmsResult as ci, ProducedProposal as cn, signManifest as cr, campaignBreakdown as ct, assertCodeSurfaceIdentity as d, PairedArmsComparison as di, SatisfiedBy as dn, neutralizationGate as dr, DiscriminationScore as dt, CrossSurfaceRelativeCost as ei, ProposalEventLike as en, sequentialPairedGate as er, SearchTaskOutcome as et, assertComponentSurface as f, PairedCorrectness as fi, TaskGold as fn, EvalFixture as fr, ScenarioSignal as ft, surfaceContentHash as g, pairRunRecords as gi, parseCorrectnessResponse as gn, EvalFixtureScenario as gr, tangleTracesRoot as gt, renderSurfaceDiff as h, pairArms as hi, createTokenRecallChecker as hn, EvalFixtureRunPlan as hr, resolveRunDir as ht, WorktreeAdapter as i, ComparePairedArmsOptions as ii, CompletionRequirement as in, SignedManifestAlgo as ir, SearchLedgerConflictError as it, SearchCandidateSlotClosedEvent as j, LabeledScenarioStoreError as jn, CrossSurfaceCandidate as jr, ProfileMatrixError as jt, SearchCandidateRegisteredEvent as k, FsLabeledScenarioStore as kn, CrossSurfaceBestSingleSelection as kr, userStoryScoreboard as kt, TransientFailureOptions as l, PairRunRecordsResult as li, ProducedState as ln, verifyManifest as lr, campaignMeanComposite as lt, componentSurfaceIdentityMaterial as m, comparePairedArms as mi, createLlmCorrectnessChecker as mn, EvalFixtureLoadOptions as mr, selectDiscriminative as mt, GitWorktreeAdapterOptions as n, CrossSurfaceSelections as ni, ToolCallEventLike as nn, HypothesisResult as nr, openSearchLedger as nt, gitWorktreeAdapter as o, MatchedRunRecordPair as oi, CorrectnessChecker as on, evaluateHypothesis as or, SearchLedgerIntegrityError as ot, codeSurfaceIdentityMaterial as p, PairedMetricDelta as pi, completionVerdict as pn, EvalFixtureFile as pr, scoreDiscrimination as pt, SearchPlannedOperation as q, SequentialDecideFn as qn, CrossSurfaceInteractionTask as qr, ProfileAxisSpec as qt, Worktree as r, CrossSurfaceTaskRow as ri, extractProducedState as rn, SignedManifest as rr, validateSearchLedgerEvent as rt, resolveWorktreePath as s, PairArmsOptions as si, LlmCorrectnessCheckerOpts as sn, hashJson as sr, CampaignBreakdown as st, CodeSurfaceVerification as t, CrossSurfaceSelectionPolicy as ti, RuntimeEventLike as tn, HypothesisManifest as tr, SearchTokenAccounting as tt, isTransientTransportFailure as u, PairedArmRow as ui, RequirementCheck as un, NeutralizationGateOptions as ur, compareRankKeys as ut, SingleRunLock as v, AnalystScenario as vi, Artifact as vn, LoadEvalFixtureScenariosOptions as vr, PlaybackDriver as vt, SearchAccountingAudit as w, composeValidators as wn, analyzeCrossSurfaceInteractions as wr, UserStoryVerdict as wt, FileSearchLedger as x, buildAnalystSurfaceDispatch as xi, ValidationIssue as xn, loadEvalFixture as xr, ScoreboardRow as xt, SingleRunLockOptions as y, BuildAnalystSurfaceDispatchOptions as yi, ArtifactValidator as yn, PlanEvalFixtureRunOptions as yr, PlaybackStep as yt, SearchLedgerEvent as z, DimensionRegression as zn, CrossSurfaceDistribution as zr, assertRealAgentReceipts as zt };
2244
+ //# sourceMappingURL=index-DE5fb3EC.d.ts.map