@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,3885 @@
1
+ import { s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
2
+ import { t as canonicalize } from "./pre-registration-DakwTRXk.js";
3
+ import { m as buildAgentProfileCell, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BuoE80Dq.js";
4
+ import { i as CostLedger } from "./cost-ledger-DIgQUFZZ.js";
5
+ import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client--GR4JbZE.js";
6
+ import { $ as SEARCH_LEDGER_FILE_CONTEXT, B as surfaceContentHash, Bt as JudgeParseError, E as pairHoldout, Et as contentHash, F as assertCodeSurfaceIdentity, J as planCampaignRun, Lt as assertRealBackend, Pt as recoverTruncatedJson, Y as runCampaign, et as SearchLedgerConflictError, h as labelTrustRank, nt as SearchLedgerIntegrityError, tt as SearchLedgerError, zt as summarizeBackendIntegrity } from "./skillopt-optimization-method-D4ODwFVV.js";
7
+ import { c as eProcess, g as mulberry32, v as pairedBootstrap } from "./statistics-CnnxdpOg.js";
8
+ import { t as comparePairedArms } from "./paired-arms-D9D0wXj2.js";
9
+ import { t as analyzeTraces } from "./analyst-LsnNpSkm.js";
10
+ import { c as campaignCellToRunRecord } from "./reward-hacking-qipEpKvY.js";
11
+ import { n as canonicalString, t as FileLedgerJournal } from "./ledger-core-DtZz1RG0.js";
12
+ import { z } from "zod";
13
+ import { closeSync, constants, existsSync, fstatSync, lstatSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, readdirSync, readlinkSync, realpathSync, rmSync, statSync, writeFileSync } from "node:fs";
14
+ import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
15
+ import { createHash, randomUUID } from "node:crypto";
16
+ import { devNull, tmpdir } from "node:os";
17
+ import { execFileSync } from "node:child_process";
18
+ import { harnessSupportsModel } from "@tangle-network/agent-interface";
19
+ //#region src/completion-verifier.ts
20
+ /**
21
+ * Completion verifier — the task-completion oracle.
22
+ *
23
+ * Answers the only eval question that is not a proxy: did the agent actually
24
+ * COMPLETE the task — produce every required deliverable, persisted and
25
+ * correct — rather than describe what should be done. A fluent transcript
26
+ * that never produces the artifact scores zero here.
27
+ *
28
+ * Per requirement, a two-stage check:
29
+ * 1. Structural — a produced item (vault artifact / approved proposal /
30
+ * tool call) of the right kind is matched against the requirement and
31
+ * carries non-empty content. Deterministic; no LLM.
32
+ * 2. Correctness — only if structurally present AND the matched item
33
+ * carries content, one targeted check decides whether that item
34
+ * actually fulfils the requirement. A hallucinated artifact fails here;
35
+ * an absent one already failed stage 1.
36
+ *
37
+ * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
38
+ * checker failures — are excluded from the denominator, never scored as
39
+ * zeros). Quality dimensions are meaningless on an incomplete task — callers
40
+ * gate on `fullyComplete` / `completionRate` before scoring quality.
41
+ */
42
+ /**
43
+ * Construct a `CompletionVerdict` from the per-requirement checks, deriving
44
+ * `completionRate` / `fullyComplete` and the spine fields (`valid` =
45
+ * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
46
+ * requirements — a verdict over nothing is a misconfiguration, mirroring
47
+ * `verifyCompletion`'s gold-spec guard.
48
+ */
49
+ function completionVerdict(input) {
50
+ if (input.requirements.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`);
51
+ const measurable = input.requirements.filter((r) => !r.unmeasured);
52
+ const unmeasuredCount = input.requirements.length - measurable.length;
53
+ if (measurable.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? "unknown reason"})`);
54
+ const satisfiedCount = measurable.filter((r) => r.satisfied).length;
55
+ const completionRate = satisfiedCount / measurable.length;
56
+ const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length;
57
+ return {
58
+ taskId: input.taskId,
59
+ requirements: input.requirements,
60
+ completionRate,
61
+ fullyComplete,
62
+ unmeasuredCount,
63
+ valid: fullyComplete,
64
+ score: completionRate
65
+ };
66
+ }
67
+ const STOPWORDS = /* @__PURE__ */ new Set([
68
+ "the",
69
+ "a",
70
+ "an",
71
+ "of",
72
+ "for",
73
+ "and",
74
+ "or",
75
+ "to",
76
+ "in",
77
+ "on",
78
+ "with",
79
+ "by"
80
+ ]);
81
+ const REQUIREMENT_FORM_STOPWORDS = /* @__PURE__ */ new Set([
82
+ "generated",
83
+ "generate",
84
+ "view",
85
+ "render",
86
+ "rendered",
87
+ "persisted",
88
+ "persist",
89
+ "artifact",
90
+ "file",
91
+ "document",
92
+ "note",
93
+ "proposal",
94
+ "deliverable",
95
+ "output",
96
+ "created",
97
+ "create",
98
+ "produce",
99
+ "produced",
100
+ "flag"
101
+ ]);
102
+ const MATCH_THRESHOLD = .5;
103
+ const MIN_CONTENT_CHARS = 50;
104
+ function tokens(s, extraStop) {
105
+ return new Set(s.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)));
106
+ }
107
+ /**
108
+ * Recall of the requirement's tokens within a candidate's identifying text.
109
+ * Recall, not Jaccard — a candidate's path/id legitimately carries extra
110
+ * tokens the requirement does not name. The requirement side drops
111
+ * deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.
112
+ */
113
+ function tokenRecall(requirementText, candidateText) {
114
+ const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS);
115
+ if (req.size === 0) return 0;
116
+ const cand = tokens(candidateText);
117
+ let hit = 0;
118
+ for (const t of req) if (cand.has(t)) hit++;
119
+ return hit / req.size;
120
+ }
121
+ function artifactCandidates(req, reqIndex, artifacts) {
122
+ const reqText = `${req.title} ${req.category ?? ""}`;
123
+ const out = [];
124
+ artifacts.forEach((a, i) => {
125
+ if ((a.content ?? "").trim().length < MIN_CONTENT_CHARS) return;
126
+ let score = tokenRecall(reqText, `${a.path ?? ""} ${a.kind} ${(a.content ?? "").slice(0, 4e3)}`);
127
+ if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) score = Math.max(score, 1);
128
+ if (score < MATCH_THRESHOLD) return;
129
+ out.push({
130
+ reqIndex,
131
+ itemKey: `artifact:${i}`,
132
+ score,
133
+ evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,
134
+ content: a.content ?? null
135
+ });
136
+ });
137
+ return out;
138
+ }
139
+ function proposalCandidates(req, reqIndex, proposals) {
140
+ const reqText = `${req.title} ${req.category ?? ""}`;
141
+ const out = [];
142
+ for (const p of proposals) {
143
+ if (p.status !== "approved") continue;
144
+ const body = (p.content ?? "").trim();
145
+ if (body.length < MIN_CONTENT_CHARS) continue;
146
+ const score = tokenRecall(reqText, `${p.title} ${body}`);
147
+ if (score < MATCH_THRESHOLD) continue;
148
+ out.push({
149
+ reqIndex,
150
+ itemKey: `proposal:${p.id}`,
151
+ score,
152
+ evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,
153
+ content: body
154
+ });
155
+ }
156
+ return out;
157
+ }
158
+ function toolCallCandidates(req, reqIndex, toolCalls) {
159
+ const out = [];
160
+ toolCalls.forEach((name, i) => {
161
+ const score = tokenRecall(req.title, name);
162
+ if (score < MATCH_THRESHOLD) return;
163
+ out.push({
164
+ reqIndex,
165
+ itemKey: `tool:${i}`,
166
+ score,
167
+ evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,
168
+ content: null
169
+ });
170
+ });
171
+ return out;
172
+ }
173
+ /**
174
+ * Verify whether a run completed the task. `checkCorrectness` is injected —
175
+ * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
176
+ *
177
+ * Throws on a gold spec with no requirements: an eval task that requires
178
+ * nothing is a misconfiguration, not a vacuously-complete task.
179
+ */
180
+ async function verifyCompletion(gold, state, checkCorrectness) {
181
+ if (gold.requirements.length === 0) throw new Error(`verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`);
182
+ const candidates = [];
183
+ gold.requirements.forEach((req, i) => {
184
+ const by = req.satisfiedBy ?? "any";
185
+ if (by === "artifact" || by === "any") candidates.push(...artifactCandidates(req, i, state.artifacts));
186
+ if (by === "proposal" || by === "any") candidates.push(...proposalCandidates(req, i, state.proposals));
187
+ if (by === "tool-call" || by === "any") candidates.push(...toolCallCandidates(req, i, state.toolCalls));
188
+ });
189
+ candidates.sort((a, b) => b.score - a.score);
190
+ const assigned = /* @__PURE__ */ new Map();
191
+ const itemTaken = /* @__PURE__ */ new Set();
192
+ for (const c of candidates) {
193
+ if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue;
194
+ assigned.set(c.reqIndex, c);
195
+ itemTaken.add(c.itemKey);
196
+ }
197
+ const requirements = [];
198
+ for (let i = 0; i < gold.requirements.length; i++) {
199
+ const req = gold.requirements[i];
200
+ const match = assigned.get(i);
201
+ const evidence = [];
202
+ let correct = null;
203
+ let unmeasuredReason;
204
+ if (match) {
205
+ evidence.push(match.evidence);
206
+ if (match.content !== null) try {
207
+ const r = await checkCorrectness(req, match.content);
208
+ correct = r.correct;
209
+ evidence.push(`correctness: ${r.correct ? "pass" : "fail"} — ${r.reason}`);
210
+ } catch (err) {
211
+ unmeasuredReason = err instanceof JudgeParseError ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}` : `checker call failed: ${err instanceof Error ? err.message : String(err)}`;
212
+ evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`);
213
+ }
214
+ else evidence.push("correctness: not assessed — matched item carries no content");
215
+ } else {
216
+ const by = req.satisfiedBy ?? "any";
217
+ const kind = by === "any" ? "artifact/proposal/tool-call" : by;
218
+ evidence.push(`no produced ${kind} matched this requirement`);
219
+ }
220
+ const structurallyPresent = match !== void 0;
221
+ const unmeasured = unmeasuredReason !== void 0;
222
+ const satisfied = structurallyPresent && !unmeasured && correct !== false;
223
+ requirements.push({
224
+ reqId: req.reqId,
225
+ title: req.title,
226
+ structurallyPresent,
227
+ correct,
228
+ satisfied,
229
+ ...unmeasured ? {
230
+ unmeasured: true,
231
+ unmeasuredReason
232
+ } : {},
233
+ evidence
234
+ });
235
+ }
236
+ return completionVerdict({
237
+ taskId: gold.taskId,
238
+ requirements
239
+ });
240
+ }
241
+ /**
242
+ * Parse the correctness checker's model response. Tolerates a response
243
+ * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
244
+ * verdict boolean usually lands in the first few tokens, so a recovered
245
+ * prefix with a boolean `correct` is a real measurement, not a guess.
246
+ * Fails loud (JudgeParseError) when no boolean verdict is recoverable.
247
+ */
248
+ function parseCorrectnessResponse(raw) {
249
+ const readVerdict = (candidate) => {
250
+ if (candidate === null || typeof candidate !== "object") return null;
251
+ const { correct, reason } = candidate;
252
+ if (typeof correct !== "boolean") return null;
253
+ return {
254
+ correct,
255
+ reason: typeof reason === "string" ? reason : ""
256
+ };
257
+ };
258
+ const match = raw.match(/\{[\s\S]*\}/);
259
+ if (match) try {
260
+ const strict = readVerdict(JSON.parse(match[0]));
261
+ if (strict) return strict;
262
+ } catch {}
263
+ const start = raw.indexOf("{");
264
+ if (start !== -1) {
265
+ const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)));
266
+ if (recovered) return recovered;
267
+ }
268
+ throw new JudgeParseError("correctness-checker", raw);
269
+ }
270
+ /**
271
+ * Production `CorrectnessChecker` — one LLM call per matched artifact,
272
+ * deterministic (temperature 0), structured JSON out. Judges fulfilment
273
+ * only: a plan, a gesture, or a description of what should be done does not
274
+ * fulfil a requirement — the artifact must BE the deliverable.
275
+ */
276
+ function createLlmCorrectnessChecker(chat, opts = {}) {
277
+ const model = opts.model ?? "claude-sonnet-4-6";
278
+ const maxContentChars = opts.maxContentChars ?? 8e3;
279
+ const maxAttempts = opts.maxAttempts ?? 2;
280
+ const costLedger = opts.costLedger ?? new CostLedger();
281
+ const sink = opts.rawSink;
282
+ const record = async (event) => {
283
+ try {
284
+ await sink?.record(event);
285
+ } catch {}
286
+ };
287
+ return async (requirement, content) => {
288
+ const request = {
289
+ model,
290
+ messages: [{
291
+ role: "system",
292
+ content: "You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}."
293
+ }, {
294
+ role: "user",
295
+ content: `Requirement: ${requirement.title}\n${requirement.category ? `Category: ${requirement.category}\n` : ""}\nProduced artifact:\n${content.slice(0, maxContentChars)}`
296
+ }],
297
+ temperature: 0,
298
+ maxTokens: 200
299
+ };
300
+ let lastErr;
301
+ for (let attempt = 0; attempt < maxAttempts; attempt++) {
302
+ const started = Date.now();
303
+ await record({
304
+ eventId: randomUUID(),
305
+ provider: chat.transport,
306
+ model,
307
+ endpoint: "/chat",
308
+ baseUrl: "",
309
+ attemptIndex: attempt,
310
+ direction: "request",
311
+ timestamp: started,
312
+ requestBody: request,
313
+ redactedFields: []
314
+ });
315
+ try {
316
+ const paid = await costLedger.runPaidCall({
317
+ channel: "verifier",
318
+ phase: opts.costPhase ?? "completion.correctness",
319
+ actor: "correctness-checker",
320
+ model,
321
+ maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
322
+ tags: {
323
+ ...opts.costTags,
324
+ requirementId: requirement.reqId,
325
+ attempt: String(attempt)
326
+ },
327
+ signal: opts.signal,
328
+ execute: (signal, callId) => chat.chat(request, {
329
+ signal,
330
+ idempotencyKey: callId
331
+ }),
332
+ receipt: costReceiptFromLlm,
333
+ receiptFromError: costReceiptFromLlmError
334
+ });
335
+ if (!paid.succeeded) throw paid.error;
336
+ const resp = paid.value;
337
+ const raw = resp.content;
338
+ await record({
339
+ eventId: randomUUID(),
340
+ provider: chat.transport,
341
+ model,
342
+ endpoint: "/chat",
343
+ baseUrl: "",
344
+ attemptIndex: attempt,
345
+ direction: "response",
346
+ timestamp: Date.now(),
347
+ durationMs: Date.now() - started,
348
+ responseBody: resp,
349
+ redactedFields: []
350
+ });
351
+ return parseCorrectnessResponse(raw);
352
+ } catch (err) {
353
+ lastErr = err;
354
+ await record({
355
+ eventId: randomUUID(),
356
+ provider: chat.transport,
357
+ model,
358
+ endpoint: "/chat",
359
+ baseUrl: "",
360
+ attemptIndex: attempt,
361
+ direction: "error",
362
+ timestamp: Date.now(),
363
+ durationMs: Date.now() - started,
364
+ errorMessage: err instanceof Error ? err.message : String(err),
365
+ redactedFields: []
366
+ });
367
+ }
368
+ }
369
+ throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
370
+ };
371
+ }
372
+ /** Stopwords for requirement-title tokenization — drops the imperative verbs
373
+ * ('review', 'update', …) common to deliverable titles so recall keys on the
374
+ * substantive nouns, not the boilerplate ask. */
375
+ const TITLE_STOPWORDS = /* @__PURE__ */ new Set([
376
+ "the",
377
+ "a",
378
+ "an",
379
+ "and",
380
+ "or",
381
+ "for",
382
+ "to",
383
+ "of",
384
+ "in",
385
+ "on",
386
+ "with",
387
+ "review",
388
+ "update",
389
+ "new",
390
+ "proposed"
391
+ ]);
392
+ /**
393
+ * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
394
+ * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
395
+ * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
396
+ * of the requirement title's significant tokens. No network.
397
+ *
398
+ * Polarity-blind: token recall credits a negation that contains the
399
+ * requirement's tokens ("I will NOT produce the comparison" recalls every token
400
+ * of "produce the comparison"). The structural match stage is ALSO lexical, so
401
+ * pairing the two collapses to a single gameable gate. Use this only as an
402
+ * opt-in structural pre-filter or for tasks whose requirements have no polarity
403
+ * to invert; for produced-state grading the correctness checker MUST be semantic
404
+ * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
405
+ */
406
+ function createTokenRecallChecker(opts = {}) {
407
+ const minRecall = opts.minRecall ?? .5;
408
+ const minLen = opts.minContentLength ?? 120;
409
+ return async (requirement, content) => {
410
+ const body = content.trim();
411
+ if (body.length < minLen) return {
412
+ correct: false,
413
+ reason: `content too thin (${body.length} chars) to be the deliverable`
414
+ };
415
+ const titleTokens = requirement.title.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t));
416
+ if (titleTokens.length === 0) return {
417
+ correct: true,
418
+ reason: "requirement title has no significant tokens — structural match accepted"
419
+ };
420
+ const lower = body.toLowerCase();
421
+ const hits = titleTokens.filter((t) => lower.includes(t)).length;
422
+ return hits / titleTokens.length >= minRecall ? {
423
+ correct: true,
424
+ reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`
425
+ } : {
426
+ correct: false,
427
+ reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`
428
+ };
429
+ };
430
+ }
431
+ //#endregion
432
+ //#region src/produced-state.ts
433
+ function artifactKind(mimeType) {
434
+ if (!mimeType) return "file";
435
+ if (mimeType.includes("json")) return "json";
436
+ if (mimeType.startsWith("text/")) return "text";
437
+ return "file";
438
+ }
439
+ /**
440
+ * Normalize a run's runtime event stream into `ProducedState`.
441
+ *
442
+ * Pure and total — unrecognized event types are skipped. `toolCalls` is
443
+ * deduplicated by name in first-seen order (completion cares about a tool's
444
+ * presence, not its call count). An artifact with neither a name nor a uri
445
+ * still yields an entry keyed by its `artifactId` so it is never silently
446
+ * dropped; an artifact with no `content` yields empty content, which the
447
+ * completion oracle's structural check then rejects on its own.
448
+ */
449
+ function extractProducedState(events) {
450
+ const artifacts = [];
451
+ const proposals = [];
452
+ const toolCalls = [];
453
+ const seenTools = /* @__PURE__ */ new Set();
454
+ for (const ev of events) if (ev.type === "tool_call") {
455
+ const name = ev.toolName;
456
+ if (name && !seenTools.has(name)) {
457
+ seenTools.add(name);
458
+ toolCalls.push(name);
459
+ }
460
+ } else if (ev.type === "artifact") {
461
+ const a = ev;
462
+ artifacts.push({
463
+ kind: artifactKind(a.mimeType),
464
+ path: a.name ?? a.uri ?? a.artifactId,
465
+ content: a.content ?? ""
466
+ });
467
+ } else if (ev.type === "proposal_created") {
468
+ const p = ev;
469
+ proposals.push({
470
+ id: p.proposalId,
471
+ title: p.title,
472
+ status: p.status ?? "pending",
473
+ ...p.content !== void 0 ? { content: p.content } : {}
474
+ });
475
+ }
476
+ return {
477
+ artifacts,
478
+ proposals,
479
+ toolCalls
480
+ };
481
+ }
482
+ //#endregion
483
+ //#region src/agent-profile.ts
484
+ /**
485
+ * The agentic coding harnesses an eval sweeps by default — the ones we care about
486
+ * ranking. This is the SINGLE source of that list; consumers import it instead of
487
+ * re-declaring their own (a re-declared list is how the fleet drifts). Pass an
488
+ * explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
489
+ * harness) to widen beyond these.
490
+ */
491
+ const CODING_HARNESSES = [
492
+ "opencode",
493
+ "claude-code",
494
+ "codex",
495
+ "kimi-code"
496
+ ];
497
+ /** Model sentinel for a vendor-locked harness that supports none of the swept models:
498
+ * it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
499
+ * resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
500
+ * model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
501
+ * table that would rot as router catalogs change. */
502
+ const HARNESS_NATIVE_MODEL = "default";
503
+ /**
504
+ * Expand a base profile across the harness × model matrix into the `AgentProfile[]`
505
+ * that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
506
+ * which models do we evaluate" lives, so no product hand-rolls its own harness list
507
+ * or column→profile mapping (the pattern that let those copies drift and silently
508
+ * break the harness pivot).
509
+ *
510
+ * Each cell clones `base`, sets `model.default`, and stamps `metadata.harness` +
511
+ * `metadata.harnessModel` (both hash-bearing, so every cell gets a distinct
512
+ * `agentProfileId` row and results join back by harness/model via {@link harnessAxisOf}
513
+ * with no hand-recomputed key). A vendor-locked harness snaps to its family's swept
514
+ * models — or its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none —
515
+ * so every requested harness runs; `keepIncompatible` forces every pair verbatim.
516
+ *
517
+ * Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
518
+ * everything we care about" switch, identical in shape whether one harness or all.
519
+ */
520
+ function expandProfileAxes(spec) {
521
+ const harnesses = spec.harnesses ?? CODING_HARNESSES;
522
+ if (harnesses.length === 0) throw new ValidationError("expandProfileAxes: no harnesses to sweep");
523
+ const baseModel = spec.base.model?.default;
524
+ const models = spec.models ?? (baseModel ? [baseModel] : []);
525
+ if (models.length === 0) throw new ValidationError("expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied");
526
+ const out = [];
527
+ const seen = /* @__PURE__ */ new Set();
528
+ for (const harness of harnesses) {
529
+ const supported = spec.keepIncompatible ? models : models.filter((model) => harnessSupportsModel(harness, model));
530
+ const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL];
531
+ for (const model of effective) {
532
+ const profile = {
533
+ ...spec.base,
534
+ name: `${spec.base.name ?? "agent"}/${harness}/${model}`,
535
+ model: {
536
+ ...spec.base.model,
537
+ default: model
538
+ },
539
+ metadata: {
540
+ ...spec.base.metadata ?? {},
541
+ harness,
542
+ harnessModel: model
543
+ }
544
+ };
545
+ const id = agentProfileId(profile);
546
+ if (seen.has(id)) continue;
547
+ seen.add(id);
548
+ out.push(profile);
549
+ }
550
+ }
551
+ if (out.length === 0) throw new ValidationError(`expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(", ")}], models=[${models.join(", ")}]).`);
552
+ return out;
553
+ }
554
+ /**
555
+ * Read the (harness, model) a matrix cell ran under, off a profile or a result row's
556
+ * profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
557
+ * wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
558
+ * this instead of recomputing an id (recomputing the wrong key is what broke the pivot
559
+ * in the hand-rolled copies).
560
+ */
561
+ function harnessAxisOf(profile) {
562
+ const m = profile.metadata;
563
+ const harness = m?.harness;
564
+ const model = m?.harnessModel;
565
+ if (typeof harness === "string" && typeof model === "string") return {
566
+ harness,
567
+ model
568
+ };
569
+ }
570
+ /**
571
+ * Collision-resistant, path-safe, human-readable profile id for eval artifacts.
572
+ * Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
573
+ * keys, and directory names where two profiles must not collapse onto one row.
574
+ * The suffix is the first 64 bits of the behaviour hash, enough for ordinary
575
+ * eval matrices while keeping filenames readable.
576
+ */
577
+ function agentProfileId(profile) {
578
+ return `${pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? "profile"}-${agentProfileHash(profile).slice(0, 16)}`;
579
+ }
580
+ /**
581
+ * Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
582
+ * model id because run records reject bare/missing model aliases.
583
+ */
584
+ function agentProfileModelId(profile) {
585
+ const model = profile.model?.default?.trim();
586
+ if (!model) throw new ValidationError(`AgentProfile "${agentProfileDisplayLabel(profile) ?? "unnamed profile"}" has no model.default — cannot record eval run`);
587
+ return model;
588
+ }
589
+ function agentProfileDisplayLabel(profile) {
590
+ return profile.name?.trim() || profile.version?.trim() || void 0;
591
+ }
592
+ function pathSafeProfileLabel(label) {
593
+ return label?.trim().replace(/[^A-Za-z0-9._-]+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "") || void 0;
594
+ }
595
+ function compact(input) {
596
+ const out = {};
597
+ for (const [key, value] of Object.entries(input)) if (value !== void 0) out[key] = value;
598
+ return out;
599
+ }
600
+ /**
601
+ * Deterministic behaviour identity for the canonical
602
+ * `@tangle-network/agent-interface` AgentProfile.
603
+ *
604
+ * `name` and `description` are labels and do not affect the hash. Profile
605
+ * `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
606
+ * and extensions do affect the hash. Resource array order is hash-bearing
607
+ * because mount order can change agent behaviour. Undefined fields are treated
608
+ * as absent; explicit `null` fields remain hash-bearing.
609
+ */
610
+ function agentProfileHash(profile) {
611
+ const model = agentProfileModelId(profile);
612
+ const behaviour = {
613
+ ...profile,
614
+ name: void 0,
615
+ description: void 0,
616
+ tags: profile.tags ? [...profile.tags].sort() : void 0,
617
+ model: compact({
618
+ ...profile.model,
619
+ default: model
620
+ })
621
+ };
622
+ return createHash("sha256").update(JSON.stringify(canonicalize(behaviour))).digest("hex");
623
+ }
624
+ //#endregion
625
+ //#region src/campaign/analyst-surface.ts
626
+ function surfaceToText(surface) {
627
+ if (typeof surface === "string") return surface;
628
+ throw new Error(`buildAnalystSurfaceDispatch: the analyst surface must be a string actorDescription, got a ${surface.kind}-tier surface. The analyst prompt is prompt-tier.`);
629
+ }
630
+ /**
631
+ * Build the `dispatchWithSurface(surface, scenario, ctx)` the improvement loop
632
+ * calls: run the analyst with `surface` as its actorDescription over the
633
+ * scenario's trace corpus and return its findings.
634
+ */
635
+ function buildAnalystSurfaceDispatch(opts) {
636
+ const analyze = opts.analyze ?? analyzeTraces;
637
+ return async (surface, scenario, _ctx) => {
638
+ const actorDescription = surfaceToText(surface);
639
+ const res = await analyze({ question: scenario.question }, {
640
+ ...opts.analystOptions,
641
+ actorDescription,
642
+ source: scenario.source
643
+ });
644
+ return {
645
+ answer: res.answer,
646
+ findings: res.findings,
647
+ actorPromptVersion: res.actorPromptVersion
648
+ };
649
+ };
650
+ }
651
+ /**
652
+ * Deterministic, ground-truth judge for analyst findings. Composite =
653
+ * recall of the scenario's `expectedFailureModes` (optionally blended with a
654
+ * precision term that penalizes findings tripping `forbiddenCues`). No LLM —
655
+ * the score is a function of the labels, so the analyst prompt is optimized
656
+ * toward surfacing real failures, not toward a judge it can flatter.
657
+ */
658
+ function failureModeRecallJudge(opts = {}) {
659
+ const recallWeight = opts.recallWeight ?? .5;
660
+ return {
661
+ name: "failure-mode-recall",
662
+ dimensions: [{
663
+ key: "recall",
664
+ description: "fraction of ground-truth failure modes the analyst surfaced"
665
+ }, {
666
+ key: "precision",
667
+ description: "1 − share of findings that named a failure/tool/error absent from this corpus"
668
+ }],
669
+ appliesTo: (s) => s.kind === "analyst-surface",
670
+ score({ artifact, scenario }) {
671
+ const modes = scenario.expectedFailureModes;
672
+ if (modes.length === 0) throw new Error(`failureModeRecallJudge: scenario '${scenario.id}' has no expectedFailureModes — refusing to score (a vacuous 1.0 would corrupt the comparison)`);
673
+ const hay = artifact.findings.join("\n").toLowerCase();
674
+ const matched = modes.filter((m) => m.cues.some((c) => hay.includes(c.toLowerCase())));
675
+ const recall = matched.length / modes.length;
676
+ const forbidden = (scenario.forbiddenCues ?? []).map((c) => c.toLowerCase());
677
+ let precision = 1;
678
+ let hallucinated = 0;
679
+ if (forbidden.length > 0) {
680
+ const denom = Math.max(1, artifact.findings.length);
681
+ hallucinated = artifact.findings.filter((f) => forbidden.some((c) => f.toLowerCase().includes(c))).length;
682
+ precision = 1 - hallucinated / denom;
683
+ }
684
+ const composite = forbidden.length > 0 ? recallWeight * recall + (1 - recallWeight) * precision : recall;
685
+ const missed = modes.filter((m) => !matched.includes(m)).map((m) => m.id);
686
+ const notes = `matched ${matched.length}/${modes.length} failure modes` + (missed.length ? `; missed [${missed.join(", ")}]` : "") + (hallucinated ? `; ${hallucinated} out-of-corpus finding(s)` : "");
687
+ return {
688
+ dimensions: {
689
+ recall,
690
+ precision
691
+ },
692
+ composite,
693
+ notes
694
+ };
695
+ }
696
+ };
697
+ }
698
+ //#endregion
699
+ //#region src/campaign/cross-surface-context.ts
700
+ function validateCrossSurfaceInput(input) {
701
+ assertNonEmptyUnique(input.taskOrder, "taskOrder");
702
+ assertNonEmptyUnique(input.componentOrder, "componentOrder");
703
+ if (input.componentOrder.length < 2) throw new ValidationError("analyzeCrossSurfaceInteractions: componentOrder must contain at least two surfaces");
704
+ assertNonEmptyUnique(input.candidateOrder, "candidateOrder");
705
+ assertNonEmptyUnique(input.costMetricOrder, "costMetricOrder");
706
+ if (input.costMetricOrder.includes("score")) throw new ValidationError(`analyzeCrossSurfaceInteractions: costMetricOrder cannot contain reserved metric 'score'`);
707
+ validateBootstrap(input);
708
+ validateSelection(input);
709
+ const componentById = indexUnique(input.components, (component) => component.componentId, "component");
710
+ assertExactSet(input.componentOrder, componentById.keys(), "componentOrder", "components");
711
+ const surfaces = /* @__PURE__ */ new Set();
712
+ for (const component of componentById.values()) {
713
+ assertNonEmpty(component.componentId, "componentId");
714
+ assertNonEmpty(component.surfaceId, `surfaceId for component '${component.componentId}'`);
715
+ if (surfaces.has(component.surfaceId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: surfaceId '${component.surfaceId}' has more than one component; the interaction stage requires one independently selected finalist per surface`);
716
+ surfaces.add(component.surfaceId);
717
+ if (typeof component.bestSingleEligible !== "boolean") throw new ValidationError(`analyzeCrossSurfaceInteractions: component '${component.componentId}' must declare bestSingleEligible`);
718
+ }
719
+ const componentIndex = new Map(input.componentOrder.map((id, index) => [id, index]));
720
+ const candidateById = indexUnique(input.candidates, (candidate) => candidate.candidateId, "candidate");
721
+ assertExactSet(input.candidateOrder, candidateById.keys(), "candidateOrder", "candidates");
722
+ if (!candidateById.has(input.baselineCandidateId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: unknown baselineCandidateId '${input.baselineCandidateId}'`);
723
+ const candidateByComponents = /* @__PURE__ */ new Map();
724
+ const singleByComponent = /* @__PURE__ */ new Map();
725
+ for (const candidate of candidateById.values()) {
726
+ validateCandidate(candidate, componentById, componentIndex, input.baselineCandidateId);
727
+ const key = crossSurfaceComponentSetKey(candidate.componentIds);
728
+ const duplicate = candidateByComponents.get(key);
729
+ if (duplicate) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidates '${duplicate.candidateId}' and '${candidate.candidateId}' materialize the same component set`);
730
+ candidateByComponents.set(key, candidate);
731
+ if (candidate.componentIds.length === 1) singleByComponent.set(candidate.componentIds[0], candidate);
732
+ }
733
+ const baseline = candidateById.get(input.baselineCandidateId);
734
+ if (baseline.componentIds.length !== 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: baseline '${baseline.candidateId}' must have zero components`);
735
+ for (const componentId of input.componentOrder) if (!singleByComponent.has(componentId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: no single-surface candidate for component '${componentId}'`);
736
+ for (let left = 0; left < input.componentOrder.length; left++) for (let right = left + 1; right < input.componentOrder.length; right++) {
737
+ const ids = [input.componentOrder[left], input.componentOrder[right]];
738
+ if (!candidateByComponents.has(crossSurfaceComponentSetKey(ids))) throw new ValidationError(`analyzeCrossSurfaceInteractions: missing pair candidate for components [${ids.join(", ")}]`);
739
+ }
740
+ const rowsByCandidate = /* @__PURE__ */ new Map();
741
+ const taskIds = new Set(input.taskOrder);
742
+ for (const row of input.rows) {
743
+ const candidate = candidateById.get(row.candidateId);
744
+ if (!candidate) throw new ValidationError(`analyzeCrossSurfaceInteractions: row for task '${row.taskId}' names unknown candidate '${row.candidateId}'`);
745
+ if (!taskIds.has(row.taskId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: row for candidate '${row.candidateId}' names task '${row.taskId}' outside the declared taskOrder`);
746
+ validateRow(input, row, candidate);
747
+ const byTask = rowsByCandidate.get(row.candidateId) ?? /* @__PURE__ */ new Map();
748
+ if (byTask.has(row.taskId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: duplicate row for candidate '${row.candidateId}' and task '${row.taskId}'`);
749
+ byTask.set(row.taskId, row);
750
+ rowsByCandidate.set(row.candidateId, byTask);
751
+ }
752
+ for (const candidateId of input.candidateOrder) {
753
+ const byTask = rowsByCandidate.get(candidateId);
754
+ const missing = input.taskOrder.filter((taskId) => !byTask?.has(taskId));
755
+ if (missing.length > 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidate '${candidateId}' is missing declared task row(s) [${missing.join(", ")}]; encode failed attempts explicitly instead of changing the task axis`);
756
+ }
757
+ const expectedRows = input.candidateOrder.length * input.taskOrder.length;
758
+ if (input.rows.length !== expectedRows) throw new ValidationError(`analyzeCrossSurfaceInteractions: expected exactly ${expectedRows} candidate × task rows, got ${input.rows.length}`);
759
+ return {
760
+ input,
761
+ components: input.componentOrder.map((id) => componentById.get(id)),
762
+ candidates: input.candidateOrder.map((id) => candidateById.get(id)),
763
+ componentById,
764
+ candidateById,
765
+ candidateByComponents,
766
+ singleByComponent,
767
+ rowsByCandidate,
768
+ componentIndex,
769
+ candidateIndex: new Map(input.candidateOrder.map((id, index) => [id, index]))
770
+ };
771
+ }
772
+ function crossSurfaceRowsFor(context, candidateId) {
773
+ return context.input.taskOrder.map((taskId) => crossSurfaceRowFor(context, candidateId, taskId));
774
+ }
775
+ function crossSurfaceRowFor(context, candidateId, taskId) {
776
+ return context.rowsByCandidate.get(candidateId).get(taskId);
777
+ }
778
+ function canonicalCrossSurfaceComponents(context, componentIds) {
779
+ return [...componentIds].sort((left, right) => context.componentIndex.get(left) - context.componentIndex.get(right));
780
+ }
781
+ function crossSurfaceComponentSetKey(componentIds) {
782
+ return JSON.stringify(componentIds);
783
+ }
784
+ function validateBootstrap(input) {
785
+ const { seed, resamples, confidence } = input.bootstrap;
786
+ if (!Number.isInteger(seed)) throw new ValidationError(`analyzeCrossSurfaceInteractions: bootstrap.seed must be an integer`);
787
+ if (!Number.isInteger(resamples) || resamples <= 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: bootstrap.resamples must be a positive integer`);
788
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`analyzeCrossSurfaceInteractions: bootstrap.confidence must be in (0,1)`);
789
+ }
790
+ function validateSelection(input) {
791
+ const policy = input.selection;
792
+ for (const [name, value] of [["minimumFiringTasks", policy.minimumFiringTasks], ["minimumEffectTasks", policy.minimumEffectTasks]]) if (!Number.isInteger(value) || value < 0 || value > input.taskOrder.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: selection.${name} must be an integer in [0,${input.taskOrder.length}]`);
793
+ if (!Number.isInteger(policy.minimumBundleComponents) || policy.minimumBundleComponents < 2 || policy.minimumBundleComponents > input.componentOrder.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: selection.minimumBundleComponents must be an integer in [2,${input.componentOrder.length}]`);
794
+ for (const [metric, limit] of Object.entries(policy.maximumMedianCostRatioToBaseline)) {
795
+ if (!input.costMetricOrder.includes(metric)) throw new ValidationError(`analyzeCrossSurfaceInteractions: cost limit names undeclared metric '${metric}'`);
796
+ if (!Number.isFinite(limit) || limit <= 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: cost ratio limit for '${metric}' must be positive and finite`);
797
+ }
798
+ }
799
+ function validateCandidate(candidate, componentById, componentIndex, baselineCandidateId) {
800
+ assertNonEmpty(candidate.candidateId, "candidateId");
801
+ assertNonEmpty(candidate.contentHash, `contentHash for candidate '${candidate.candidateId}'`);
802
+ if (!Number.isInteger(candidate.artifactBytes) || candidate.artifactBytes < 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: artifactBytes for candidate '${candidate.candidateId}' must be a non-negative integer`);
803
+ assertUnique$1(candidate.componentIds, `componentIds for candidate '${candidate.candidateId}'`);
804
+ for (const componentId of candidate.componentIds) if (!componentById.has(componentId)) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidate '${candidate.candidateId}' names unknown component '${componentId}'`);
805
+ const canonical = [...candidate.componentIds].sort((left, right) => componentIndex.get(left) - componentIndex.get(right));
806
+ if (!arraysEqual(candidate.componentIds, canonical)) throw new ValidationError(`analyzeCrossSurfaceInteractions: candidate '${candidate.candidateId}' components must follow componentOrder`);
807
+ if (candidate.candidateId !== baselineCandidateId && candidate.componentIds.length === 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: only baseline '${baselineCandidateId}' may have zero components`);
808
+ }
809
+ function validateRow(input, row, candidate) {
810
+ if (!arraysEqual(row.componentIds, candidate.componentIds)) throw new ValidationError(`analyzeCrossSurfaceInteractions: row '${row.candidateId}/${row.taskId}' componentIds do not match its candidate`);
811
+ if (![
812
+ "complete",
813
+ "missing",
814
+ "invalid"
815
+ ].includes(row.completeness)) throw new ValidationError(`analyzeCrossSurfaceInteractions: row '${row.candidateId}/${row.taskId}' has unknown completeness '${String(row.completeness)}'`);
816
+ if (row.completeness === "complete") {
817
+ if (typeof row.pass !== "boolean" || !Number.isFinite(row.score)) throw new ValidationError(`analyzeCrossSurfaceInteractions: complete row '${row.candidateId}/${row.taskId}' requires boolean pass and finite score`);
818
+ if (row.rejectReason !== null) throw new ValidationError(`analyzeCrossSurfaceInteractions: complete row '${row.candidateId}/${row.taskId}' cannot carry rejectReason`);
819
+ } else {
820
+ if (row.pass !== null || row.score !== null) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${row.completeness} row '${row.candidateId}/${row.taskId}' must use null pass and score`);
821
+ if (typeof row.rejectReason !== "string" || row.rejectReason.trim() === "") throw new ValidationError(`analyzeCrossSurfaceInteractions: ${row.completeness} row '${row.candidateId}/${row.taskId}' requires rejectReason`);
822
+ }
823
+ assertExactSet(input.costMetricOrder, Object.keys(row.cost), `cost keys for row '${row.candidateId}/${row.taskId}'`, "costMetricOrder");
824
+ for (const metric of input.costMetricOrder) {
825
+ const value = row.cost[metric];
826
+ if (value === null || value === void 0 || !Number.isFinite(value) || value < 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: unknown or invalid cost '${metric}' on row '${row.candidateId}/${row.taskId}'; every attempt must report a non-negative finite value`);
827
+ }
828
+ const evidenceByComponent = indexUnique(row.componentEvidence, (evidence) => evidence.componentId, `componentEvidence on row '${row.candidateId}/${row.taskId}'`);
829
+ assertExactSet(candidate.componentIds, evidenceByComponent.keys(), `componentEvidence on row '${row.candidateId}/${row.taskId}'`, "candidate components");
830
+ for (const evidence of evidenceByComponent.values()) {
831
+ assertTriState(evidence.fired, "fired", row);
832
+ assertTriState(evidence.effectObserved, "effectObserved", row);
833
+ if (evidence.effectObserved === true && evidence.fired !== true) throw new ValidationError(`analyzeCrossSurfaceInteractions: effectObserved=true requires fired=true for component '${evidence.componentId}' on row '${row.candidateId}/${row.taskId}'`);
834
+ }
835
+ }
836
+ function assertTriState(value, field, row) {
837
+ if (value !== true && value !== false && value !== null) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${field} on row '${row.candidateId}/${row.taskId}' must be boolean or null`);
838
+ }
839
+ function indexUnique(items, id, label) {
840
+ const result = /* @__PURE__ */ new Map();
841
+ for (const item of items) {
842
+ const key = id(item);
843
+ assertNonEmpty(key, `${label} id`);
844
+ if (result.has(key)) throw new ValidationError(`analyzeCrossSurfaceInteractions: duplicate ${label} id '${key}'`);
845
+ result.set(key, item);
846
+ }
847
+ return result;
848
+ }
849
+ function assertNonEmptyUnique(values, label) {
850
+ if (values.length === 0) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${label} is empty`);
851
+ assertUnique$1(values, label);
852
+ for (const value of values) assertNonEmpty(value, label);
853
+ }
854
+ function assertUnique$1(values, label) {
855
+ if (new Set(values).size !== values.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${label} contains duplicates`);
856
+ }
857
+ function assertNonEmpty(value, label) {
858
+ if (typeof value !== "string" || value.trim() === "") throw new ValidationError(`analyzeCrossSurfaceInteractions: ${label} must be a non-empty string`);
859
+ }
860
+ function assertExactSet(expected, actualIterable, actualLabel, expectedLabel) {
861
+ const actual = [...actualIterable];
862
+ const expectedSet = new Set(expected);
863
+ const actualSet = new Set(actual);
864
+ const missing = expected.filter((value) => !actualSet.has(value));
865
+ const extra = actual.filter((value) => !expectedSet.has(value));
866
+ if (missing.length > 0 || extra.length > 0 || actual.length !== expected.length) throw new ValidationError(`analyzeCrossSurfaceInteractions: ${actualLabel} does not match ${expectedLabel}; missing=[${missing.join(", ")}], extra=[${extra.join(", ")}]`);
867
+ }
868
+ function arraysEqual(left, right) {
869
+ return left.length === right.length && left.every((value, index) => value === right[index]);
870
+ }
871
+ //#endregion
872
+ //#region src/campaign/cross-surface-interaction.ts
873
+ /**
874
+ * Task-paired comparison and deterministic selection for independently
875
+ * proposed agent surfaces. Unlike factorial cell-mean attribution, every
876
+ * benefit, regression, and interaction remains attached to its task row.
877
+ */
878
+ /**
879
+ * Build the complete cross-surface evidence matrix and derive all three frozen
880
+ * candidates. The task/candidate/component orders are part of the input so
881
+ * neither insertion order nor an after-the-fact tie-break can change a result.
882
+ */
883
+ function analyzeCrossSurfaceInteractions(input) {
884
+ const context = validateCrossSurfaceInput(input);
885
+ const baseline = context.candidateById.get(input.baselineCandidateId);
886
+ const preliminary = context.candidates.map((candidate) => summarizeCandidate(context, candidate, baseline));
887
+ const baselineSummary = preliminary.find((summary) => summary.candidate.candidateId === input.baselineCandidateId);
888
+ const summaries = preliminary.map((summary) => summary.candidate.candidateId === input.baselineCandidateId ? summary : {
889
+ ...summary,
890
+ eligibility: candidateEligibility(context, summary, baselineSummary)
891
+ });
892
+ const summaryById = new Map(summaries.map((summary) => [summary.candidate.candidateId, summary]));
893
+ const eligibleSingles = rankedEligibleSingles(context, summaryById);
894
+ const interactionReadySingles = rankedInteractionReadySingles(context, summaryById);
895
+ const pairwise = buildPairwise(context, summaryById, interactionReadySingles, baselineSummary);
896
+ const selections = selectCandidates(context, summaryById, eligibleSingles, interactionReadySingles, pairwise);
897
+ const rows = context.candidates.flatMap((candidate) => input.taskOrder.map((taskId) => context.rowsByCandidate.get(candidate.candidateId).get(taskId)));
898
+ return {
899
+ taskIds: [...input.taskOrder],
900
+ componentIds: [...input.componentOrder],
901
+ candidateIds: [...input.candidateOrder],
902
+ costMetrics: [...input.costMetricOrder],
903
+ rows,
904
+ missingAttempts: rows.filter((row) => row.completeness === "missing"),
905
+ invalidAttempts: rows.filter((row) => row.completeness === "invalid"),
906
+ candidates: summaries,
907
+ pairwise,
908
+ selections
909
+ };
910
+ }
911
+ function summarizeCandidate(context, candidate, baseline) {
912
+ const rows = crossSurfaceRowsFor(context, candidate.candidateId);
913
+ const outcome = summarizeOutcome(context, rows, crossSurfaceRowsFor(context, baseline.candidateId), candidate.candidateId === baseline.candidateId);
914
+ const completeScores = rows.filter((row) => row.completeness === "complete").map((row) => row.score);
915
+ const costs = Object.fromEntries(context.input.costMetricOrder.map((metric) => [metric, distribution(rows.map((row) => row.cost[metric]))]));
916
+ return {
917
+ candidate,
918
+ outcome,
919
+ score: completeScores.length === 0 ? null : distribution(completeScores),
920
+ costs,
921
+ firing: summarizeCandidateEvidence(rows, candidate.componentIds, "fired"),
922
+ effect: summarizeCandidateEvidence(rows, candidate.componentIds, "effectObserved"),
923
+ comparisonToBaseline: candidate.candidateId === baseline.candidateId ? null : pairedComparison(context, baseline.candidateId, candidate.candidateId),
924
+ eligibility: null
925
+ };
926
+ }
927
+ function summarizeOutcome(context, rows, baselineRows, isBaseline) {
928
+ const resolvedTaskIds = [];
929
+ const failedTaskIds = [];
930
+ const missingTaskIds = [];
931
+ const invalidTaskIds = [];
932
+ const benefitTaskIds = [];
933
+ const regressionTaskIds = [];
934
+ const comparisonMissingTaskIds = [];
935
+ for (let index = 0; index < context.input.taskOrder.length; index++) {
936
+ const row = rows[index];
937
+ const baseline = baselineRows[index];
938
+ if (row.completeness === "missing") missingTaskIds.push(row.taskId);
939
+ else if (row.completeness === "invalid") invalidTaskIds.push(row.taskId);
940
+ else if (row.pass) resolvedTaskIds.push(row.taskId);
941
+ else failedTaskIds.push(row.taskId);
942
+ if (isBaseline) continue;
943
+ if (row.completeness !== "complete" || baseline.completeness !== "complete") comparisonMissingTaskIds.push(row.taskId);
944
+ else if (row.pass && !baseline.pass) benefitTaskIds.push(row.taskId);
945
+ else if (!row.pass && baseline.pass) regressionTaskIds.push(row.taskId);
946
+ }
947
+ return {
948
+ resolvedTaskIds,
949
+ failedTaskIds,
950
+ missingTaskIds,
951
+ invalidTaskIds,
952
+ benefitTaskIds,
953
+ regressionTaskIds,
954
+ comparisonMissingTaskIds,
955
+ netBenefit: benefitTaskIds.length - regressionTaskIds.length
956
+ };
957
+ }
958
+ function summarizeCandidateEvidence(rows, componentIds, field) {
959
+ if (componentIds.length === 0) return {
960
+ byComponent: [],
961
+ allObservedTaskIds: [],
962
+ someObservedTaskIds: [],
963
+ noneObservedTaskIds: [],
964
+ unobservedTaskIds: []
965
+ };
966
+ const byComponent = componentIds.map((componentId) => {
967
+ const observedTaskIds = [];
968
+ const notObservedTaskIds = [];
969
+ const unobservedTaskIds = [];
970
+ for (const row of rows) {
971
+ const value = evidenceValue(row, componentId, field);
972
+ if (value === true) observedTaskIds.push(row.taskId);
973
+ else if (value === false) notObservedTaskIds.push(row.taskId);
974
+ else unobservedTaskIds.push(row.taskId);
975
+ }
976
+ return {
977
+ componentId,
978
+ observedTaskIds,
979
+ notObservedTaskIds,
980
+ unobservedTaskIds
981
+ };
982
+ });
983
+ const allObservedTaskIds = [];
984
+ const someObservedTaskIds = [];
985
+ const noneObservedTaskIds = [];
986
+ const unobservedTaskIds = [];
987
+ for (const row of rows) {
988
+ const values = componentIds.map((componentId) => evidenceValue(row, componentId, field));
989
+ if (values.some((value) => value === null)) unobservedTaskIds.push(row.taskId);
990
+ else if (values.every(Boolean)) allObservedTaskIds.push(row.taskId);
991
+ else if (values.some(Boolean)) someObservedTaskIds.push(row.taskId);
992
+ else noneObservedTaskIds.push(row.taskId);
993
+ }
994
+ return {
995
+ byComponent,
996
+ allObservedTaskIds,
997
+ someObservedTaskIds,
998
+ noneObservedTaskIds,
999
+ unobservedTaskIds
1000
+ };
1001
+ }
1002
+ function candidateEligibility(context, summary, baseline) {
1003
+ const reasons = [];
1004
+ if (summary.outcome.missingTaskIds.length > 0) reasons.push("missing_attempt");
1005
+ if (summary.outcome.invalidTaskIds.length > 0) reasons.push("invalid_attempt");
1006
+ if (summary.outcome.comparisonMissingTaskIds.length > 0) reasons.push("baseline_outcome_missing");
1007
+ if (summary.outcome.benefitTaskIds.length <= summary.outcome.regressionTaskIds.length) reasons.push("benefit_not_greater_than_regression");
1008
+ appendEvidenceEligibilityReasons(reasons, summary.firing, context.input.selection.minimumFiringTasks, context.input.selection.requireObservedFiring, "firing_below_minimum", "firing_unobserved");
1009
+ appendEvidenceEligibilityReasons(reasons, summary.effect, context.input.selection.minimumEffectTasks, context.input.selection.requireObservedEffect, "effect_below_minimum", "effect_unobserved");
1010
+ if (!withinCostLimits(context, summary, baseline)) reasons.push("cost_limit_exceeded");
1011
+ return {
1012
+ eligible: reasons.length === 0,
1013
+ reasons: unique(reasons)
1014
+ };
1015
+ }
1016
+ function appendEvidenceEligibilityReasons(reasons, evidence, minimum, requireObserved, belowReason, unobservedReason) {
1017
+ if (evidence.byComponent.some((component) => component.observedTaskIds.length < minimum)) reasons.push(belowReason);
1018
+ if (requireObserved && evidence.byComponent.some((component) => component.unobservedTaskIds.length > 0)) reasons.push(unobservedReason);
1019
+ }
1020
+ function rankedEligibleSingles(context, summaryById) {
1021
+ return [...context.singleByComponent.values()].map((candidate) => summaryById.get(candidate.candidateId)).filter((summary) => summary.eligibility?.eligible).sort((left, right) => compareSingleSummaries(context, left, right));
1022
+ }
1023
+ /**
1024
+ * A neutral constituent may seed a composition when it has complete, bounded,
1025
+ * observed evidence and causes no baseline regression. Individual benefit is
1026
+ * deliberately left to the best-single arm rather than used as a pair gate.
1027
+ */
1028
+ function rankedInteractionReadySingles(context, summaryById) {
1029
+ return [...context.singleByComponent.values()].map((candidate) => summaryById.get(candidate.candidateId)).filter((summary) => summary.outcome.regressionTaskIds.length === 0 && summary.eligibility?.reasons.every((reason) => reason === "benefit_not_greater_than_regression")).sort((left, right) => compareSingleSummaries(context, left, right));
1030
+ }
1031
+ function compareSingleSummaries(context, left, right) {
1032
+ const byNetBenefit = right.outcome.netBenefit - left.outcome.netBenefit;
1033
+ if (byNetBenefit !== 0) return byNetBenefit;
1034
+ const byBenefit = right.outcome.benefitTaskIds.length - left.outcome.benefitTaskIds.length;
1035
+ if (byBenefit !== 0) return byBenefit;
1036
+ const byRegression = left.outcome.regressionTaskIds.length - right.outcome.regressionTaskIds.length;
1037
+ if (byRegression !== 0) return byRegression;
1038
+ for (const metric of context.input.costMetricOrder) {
1039
+ const byCost = left.costs[metric].median - right.costs[metric].median;
1040
+ if (byCost !== 0) return byCost;
1041
+ }
1042
+ const byBytes = left.candidate.artifactBytes - right.candidate.artifactBytes;
1043
+ if (byBytes !== 0) return byBytes;
1044
+ return context.candidateIndex.get(left.candidate.candidateId) - context.candidateIndex.get(right.candidate.candidateId);
1045
+ }
1046
+ function buildPairwise(context, summaryById, interactionReadySingles, baseline) {
1047
+ const interactionReadyIds = new Set(interactionReadySingles.map((summary) => summary.candidate.candidateId));
1048
+ const entries = [];
1049
+ for (let leftIndex = 0; leftIndex < context.components.length; leftIndex++) for (let rightIndex = leftIndex + 1; rightIndex < context.components.length; rightIndex++) {
1050
+ const leftComponent = context.components[leftIndex];
1051
+ const rightComponent = context.components[rightIndex];
1052
+ const leftSingle = context.singleByComponent.get(leftComponent.componentId);
1053
+ const rightSingle = context.singleByComponent.get(rightComponent.componentId);
1054
+ const pair = context.candidateByComponents.get(crossSurfaceComponentSetKey([leftComponent.componentId, rightComponent.componentId]));
1055
+ const pairSummary = summaryById.get(pair.candidateId);
1056
+ const leftSummary = summaryById.get(leftSingle.candidateId);
1057
+ const rightSummary = summaryById.get(rightSingle.candidateId);
1058
+ const synergyTaskIds = [];
1059
+ const interferenceTaskIds = [];
1060
+ for (const taskId of context.input.taskOrder) {
1061
+ const pairRow = crossSurfaceRowFor(context, pair.candidateId, taskId);
1062
+ const leftRow = crossSurfaceRowFor(context, leftSingle.candidateId, taskId);
1063
+ const rightRow = crossSurfaceRowFor(context, rightSingle.candidateId, taskId);
1064
+ if (![
1065
+ pairRow,
1066
+ leftRow,
1067
+ rightRow
1068
+ ].every((row) => row.completeness === "complete")) continue;
1069
+ if (pairRow.pass && !leftRow.pass && !rightRow.pass) synergyTaskIds.push(taskId);
1070
+ if (!pairRow.pass && (leftRow.pass || rightRow.pass)) interferenceTaskIds.push(taskId);
1071
+ }
1072
+ const incrementalVsConstituents = [compareCandidates(context, summaryById, leftSingle.candidateId, pair.candidateId), compareCandidates(context, summaryById, rightSingle.candidateId, pair.candidateId)];
1073
+ const betterSingle = compareSingleSummaries(context, leftSummary, rightSummary) <= 0 ? leftSummary : rightSummary;
1074
+ const firing = summarizePairEvidence(crossSurfaceRowsFor(context, pair.candidateId), leftComponent.componentId, rightComponent.componentId, "fired");
1075
+ const effect = summarizePairEvidence(crossSurfaceRowsFor(context, pair.candidateId), leftComponent.componentId, rightComponent.componentId, "effectObserved");
1076
+ const compatibility = pairCompatibility(context, pairSummary, baseline, betterSingle, incrementalVsConstituents, interactionReadyIds.has(leftSingle.candidateId) && interactionReadyIds.has(rightSingle.candidateId), interferenceTaskIds, firing, effect);
1077
+ entries.push({
1078
+ componentIds: [leftComponent.componentId, rightComponent.componentId],
1079
+ singleCandidateIds: [leftSingle.candidateId, rightSingle.candidateId],
1080
+ compositionCandidateId: pair.candidateId,
1081
+ benefitTaskIds: [...pairSummary.outcome.benefitTaskIds],
1082
+ regressionTaskIds: [...pairSummary.outcome.regressionTaskIds],
1083
+ synergyTaskIds,
1084
+ interferenceTaskIds,
1085
+ incrementalVsConstituents,
1086
+ relativeCostToBaseline: relativeCosts(context, pairSummary, baseline),
1087
+ firing,
1088
+ effect,
1089
+ interaction: interactionEffect(context, baseline.candidate, [leftSingle, rightSingle], pair),
1090
+ compatibility
1091
+ });
1092
+ }
1093
+ return entries;
1094
+ }
1095
+ function pairCompatibility(context, pair, baseline, betterSingle, comparisons, constituentsReady, interferenceTaskIds, firing, effect) {
1096
+ const reasons = [];
1097
+ if (!constituentsReady) reasons.push("constituent_not_ready");
1098
+ if (pair.outcome.missingTaskIds.length > 0 || pair.outcome.invalidTaskIds.length > 0 || pair.outcome.comparisonMissingTaskIds.length > 0) reasons.push("pair_incomplete");
1099
+ if (pair.outcome.regressionTaskIds.length > 0) reasons.push("baseline_regression");
1100
+ if (interferenceTaskIds.length > 0) reasons.push("interference");
1101
+ if (comparisons.find((comparison) => comparison.comparatorCandidateId === betterSingle.candidate.candidateId).winsTaskIds.length === 0) reasons.push("no_incremental_resolution");
1102
+ appendPairEvidenceReasons(reasons, firing, context.input.selection.minimumFiringTasks, context.input.selection.requireObservedFiring, "firing_below_minimum", "firing_unobserved");
1103
+ appendPairEvidenceReasons(reasons, effect, context.input.selection.minimumEffectTasks, context.input.selection.requireObservedEffect, "effect_below_minimum", "effect_unobserved");
1104
+ if (!withinCostLimits(context, pair, baseline)) reasons.push("cost_limit_exceeded");
1105
+ return {
1106
+ compatible: reasons.length === 0,
1107
+ reasons: unique(reasons),
1108
+ betterSingleCandidateId: betterSingle.candidate.candidateId
1109
+ };
1110
+ }
1111
+ function appendPairEvidenceReasons(reasons, evidence, minimum, requireObserved, belowReason, unobservedReason) {
1112
+ const leftObserved = evidence.bothTaskIds.length + evidence.leftOnlyTaskIds.length;
1113
+ const rightObserved = evidence.bothTaskIds.length + evidence.rightOnlyTaskIds.length;
1114
+ if (leftObserved < minimum || rightObserved < minimum) reasons.push(belowReason);
1115
+ if (requireObserved && evidence.unobservedTaskIds.length > 0) reasons.push(unobservedReason);
1116
+ }
1117
+ function interactionEffect(context, baseline, singles, composition) {
1118
+ const perTask = context.input.taskOrder.map((taskId) => {
1119
+ const baselineRow = crossSurfaceRowFor(context, baseline.candidateId, taskId);
1120
+ const singleRows = singles.map((single) => crossSurfaceRowFor(context, single.candidateId, taskId));
1121
+ const compositionRow = crossSurfaceRowFor(context, composition.candidateId, taskId);
1122
+ if (![
1123
+ baselineRow,
1124
+ ...singleRows,
1125
+ compositionRow
1126
+ ].every((row) => row.completeness === "complete")) return {
1127
+ taskId,
1128
+ passInteraction: null,
1129
+ scoreInteraction: null
1130
+ };
1131
+ const additivePass = singleRows.reduce((sum, row) => sum + Number(row.pass), 0) - (singles.length - 1) * Number(baselineRow.pass);
1132
+ const additiveScore = singleRows.reduce((sum, row) => sum + row.score, 0) - (singles.length - 1) * baselineRow.score;
1133
+ return {
1134
+ taskId,
1135
+ passInteraction: Number(compositionRow.pass) - additivePass,
1136
+ scoreInteraction: compositionRow.score - additiveScore
1137
+ };
1138
+ });
1139
+ const complete = perTask.filter((row) => row.passInteraction !== null && row.scoreInteraction !== null);
1140
+ const passInteractions = complete.map((row) => row.passInteraction);
1141
+ const scoreInteractions = complete.map((row) => row.scoreInteraction);
1142
+ return {
1143
+ perTask,
1144
+ n: complete.length,
1145
+ nMissing: perTask.length - complete.length,
1146
+ meanPassInteraction: meanOrNull$1(passInteractions),
1147
+ meanScoreInteraction: meanOrNull$1(scoreInteractions),
1148
+ passBootstrap: bootstrapInteraction(passInteractions, context),
1149
+ scoreBootstrap: bootstrapInteraction(scoreInteractions, context)
1150
+ };
1151
+ }
1152
+ function bootstrapInteraction(interactions, context) {
1153
+ if (interactions.length === 0) return null;
1154
+ return pairedBootstrap(new Array(interactions.length).fill(0), interactions, {
1155
+ ...context.input.bootstrap,
1156
+ statistic: "mean"
1157
+ });
1158
+ }
1159
+ function selectCandidates(context, summaryById, eligibleSingles, interactionReadySingles, pairwise) {
1160
+ const bestSingleRanking = eligibleSingles.filter((summary) => {
1161
+ const componentId = summary.candidate.componentIds[0];
1162
+ return context.componentById.get(componentId).bestSingleEligible;
1163
+ });
1164
+ const ranking = bestSingleRanking.map((summary, index) => ({
1165
+ rank: index + 1,
1166
+ candidateId: summary.candidate.candidateId,
1167
+ componentId: summary.candidate.componentIds[0]
1168
+ }));
1169
+ const best = bestSingleRanking[0];
1170
+ return {
1171
+ bestSingle: best ? {
1172
+ candidateId: best.candidate.candidateId,
1173
+ componentId: best.candidate.componentIds[0],
1174
+ ranking
1175
+ } : null,
1176
+ naiveStack: selectNaiveStack(context, eligibleSingles),
1177
+ interactionAware: selectInteractionAware(context, summaryById, interactionReadySingles, pairwise)
1178
+ };
1179
+ }
1180
+ function selectNaiveStack(context, eligibleSingles) {
1181
+ if (eligibleSingles.length === 0) return null;
1182
+ const componentIds = eligibleSingles.map((summary) => summary.candidate.componentIds[0]).sort((left, right) => context.componentIndex.get(left) - context.componentIndex.get(right));
1183
+ const candidate = context.candidateByComponents.get(crossSurfaceComponentSetKey(componentIds));
1184
+ if (!candidate) throw new ValidationError(`analyzeCrossSurfaceInteractions: naive stack [${componentIds.join(", ")}] was not evaluated`);
1185
+ return {
1186
+ candidateId: candidate.candidateId,
1187
+ componentIds
1188
+ };
1189
+ }
1190
+ function selectInteractionAware(context, summaryById, interactionReadySingles, pairwise) {
1191
+ const baseline = summaryById.get(context.input.baselineCandidateId);
1192
+ const pairByComponents = new Map(pairwise.map((entry) => [crossSurfaceComponentSetKey(entry.componentIds), entry]));
1193
+ const paths = pairwise.filter((entry) => entry.compatibility.compatible).map((entry) => growInteractionPath(context, summaryById, interactionReadySingles, pairByComponents, baseline, summaryById.get(entry.compositionCandidateId)));
1194
+ if (paths.length === 0) return null;
1195
+ const qualified = paths.filter((path) => path.qualified).sort((left, right) => compareInteractionPaths(context, summaryById, left, right));
1196
+ const winning = qualified[0] ?? [...paths].sort((left, right) => compareInteractionPaths(context, summaryById, left, right))[0];
1197
+ return {
1198
+ seedCandidateId: winning.seedCandidateId,
1199
+ terminalCandidateId: winning.terminalCandidateId,
1200
+ terminalComponentIds: [...winning.terminalComponentIds],
1201
+ selectedCandidateId: qualified.length > 0 ? winning.terminalCandidateId : null,
1202
+ qualified: qualified.length > 0,
1203
+ evaluatedPaths: paths,
1204
+ steps: winning.steps
1205
+ };
1206
+ }
1207
+ function growInteractionPath(context, summaryById, interactionReadySingles, pairByComponents, baseline, seed) {
1208
+ let current = seed;
1209
+ let retained = [...seed.candidate.componentIds];
1210
+ let remaining = interactionReadySingles.filter((summary) => !retained.includes(summary.candidate.componentIds[0]));
1211
+ const steps = [];
1212
+ while (remaining.length > 0) {
1213
+ const decisions = remaining.map((addition) => evaluateAddition(context, summaryById, pairByComponents, baseline, current, retained, addition));
1214
+ const selected = decisions.filter((decision) => decision.eligible).sort((left, right) => compareAdditionDecisions(context, summaryById, left, right))[0];
1215
+ if (selected) selected.selected = true;
1216
+ steps.push({
1217
+ fromCandidateId: current.candidate.candidateId,
1218
+ retainedComponentIds: [...retained],
1219
+ considered: decisions.sort((left, right) => context.candidateIndex.get(left.additionCandidateId) - context.candidateIndex.get(right.additionCandidateId)),
1220
+ selectedCandidateId: selected?.bundleCandidateId ?? null
1221
+ });
1222
+ if (!selected?.bundleCandidateId) break;
1223
+ current = summaryById.get(selected.bundleCandidateId);
1224
+ retained = [...current.candidate.componentIds];
1225
+ remaining = remaining.filter((summary) => summary.candidate.candidateId !== selected.additionCandidateId);
1226
+ }
1227
+ const qualified = retained.length >= context.input.selection.minimumBundleComponents;
1228
+ return {
1229
+ seedCandidateId: seed.candidate.candidateId,
1230
+ terminalCandidateId: current.candidate.candidateId,
1231
+ terminalComponentIds: retained,
1232
+ qualified,
1233
+ steps
1234
+ };
1235
+ }
1236
+ function compareInteractionPaths(context, summaryById, left, right) {
1237
+ const leftSummary = summaryById.get(left.terminalCandidateId);
1238
+ const rightSummary = summaryById.get(right.terminalCandidateId);
1239
+ const byNetBenefit = rightSummary.outcome.netBenefit - leftSummary.outcome.netBenefit;
1240
+ if (byNetBenefit !== 0) return byNetBenefit;
1241
+ const byBenefit = rightSummary.outcome.benefitTaskIds.length - leftSummary.outcome.benefitTaskIds.length;
1242
+ if (byBenefit !== 0) return byBenefit;
1243
+ const byRegression = leftSummary.outcome.regressionTaskIds.length - rightSummary.outcome.regressionTaskIds.length;
1244
+ if (byRegression !== 0) return byRegression;
1245
+ for (const metric of context.input.costMetricOrder) {
1246
+ const byCost = leftSummary.costs[metric].median - rightSummary.costs[metric].median;
1247
+ if (byCost !== 0) return byCost;
1248
+ }
1249
+ const byBytes = leftSummary.candidate.artifactBytes - rightSummary.candidate.artifactBytes;
1250
+ if (byBytes !== 0) return byBytes;
1251
+ const byCandidateOrder = context.candidateIndex.get(left.terminalCandidateId) - context.candidateIndex.get(right.terminalCandidateId);
1252
+ if (byCandidateOrder !== 0) return byCandidateOrder;
1253
+ return context.candidateIndex.get(left.seedCandidateId) - context.candidateIndex.get(right.seedCandidateId);
1254
+ }
1255
+ function evaluateAddition(context, summaryById, pairByComponents, baseline, current, retained, addition) {
1256
+ const additionComponentId = addition.candidate.componentIds[0];
1257
+ const reasons = [];
1258
+ for (const retainedComponentId of retained) if (!pairByComponents.get(crossSurfaceComponentSetKey(canonicalCrossSurfaceComponents(context, [retainedComponentId, additionComponentId])))?.compatibility.compatible) reasons.push("pair_incompatible");
1259
+ const bundleComponents = canonicalCrossSurfaceComponents(context, [...retained, additionComponentId]);
1260
+ const bundle = context.candidateByComponents.get(crossSurfaceComponentSetKey(bundleComponents));
1261
+ if (!bundle) {
1262
+ reasons.push("full_bundle_not_evaluated");
1263
+ return emptyAdditionDecision(addition, additionComponentId, reasons);
1264
+ }
1265
+ const bundleSummary = summaryById.get(bundle.candidateId);
1266
+ if (bundleSummary.outcome.missingTaskIds.length > 0 || bundleSummary.outcome.invalidTaskIds.length > 0 || bundleSummary.outcome.comparisonMissingTaskIds.length > 0) reasons.push("bundle_incomplete");
1267
+ if (bundleSummary.outcome.regressionTaskIds.length > 0) reasons.push("baseline_regression");
1268
+ const comparison = compareCandidates(context, summaryById, current.candidate.candidateId, bundle.candidateId);
1269
+ if (comparison.winsTaskIds.length === 0) reasons.push("no_incremental_resolution");
1270
+ if (comparison.regressionTaskIds.length > 0) reasons.push("incremental_regression");
1271
+ appendBundleEvidenceReasons(reasons, bundleSummary.firing, context.input.selection.minimumFiringTasks, context.input.selection.requireObservedFiring, "firing_below_minimum", "firing_unobserved");
1272
+ appendBundleEvidenceReasons(reasons, bundleSummary.effect, context.input.selection.minimumEffectTasks, context.input.selection.requireObservedEffect, "effect_below_minimum", "effect_unobserved");
1273
+ if (!withinCostLimits(context, bundleSummary, baseline)) reasons.push("cost_limit_exceeded");
1274
+ return {
1275
+ additionCandidateId: addition.candidate.candidateId,
1276
+ additionComponentId,
1277
+ bundleCandidateId: bundle.candidateId,
1278
+ incrementalResolutionTaskIds: comparison.winsTaskIds,
1279
+ incrementalRegressionTaskIds: comparison.regressionTaskIds,
1280
+ incrementalMedianCost: Object.fromEntries(context.input.costMetricOrder.map((metric) => [metric, bundleSummary.costs[metric].median - current.costs[metric].median])),
1281
+ eligible: reasons.length === 0,
1282
+ selected: false,
1283
+ reasons: unique(reasons)
1284
+ };
1285
+ }
1286
+ function emptyAdditionDecision(addition, additionComponentId, reasons) {
1287
+ return {
1288
+ additionCandidateId: addition.candidate.candidateId,
1289
+ additionComponentId,
1290
+ bundleCandidateId: null,
1291
+ incrementalResolutionTaskIds: [],
1292
+ incrementalRegressionTaskIds: [],
1293
+ incrementalMedianCost: null,
1294
+ eligible: false,
1295
+ selected: false,
1296
+ reasons: unique(reasons)
1297
+ };
1298
+ }
1299
+ function appendBundleEvidenceReasons(reasons, evidence, minimum, requireObserved, belowReason, unobservedReason) {
1300
+ if (evidence.byComponent.some((component) => component.observedTaskIds.length < minimum)) reasons.push(belowReason);
1301
+ if (requireObserved && evidence.byComponent.some((component) => component.unobservedTaskIds.length > 0)) reasons.push(unobservedReason);
1302
+ }
1303
+ function compareAdditionDecisions(context, summaryById, left, right) {
1304
+ const byWins = right.incrementalResolutionTaskIds.length - left.incrementalResolutionTaskIds.length;
1305
+ if (byWins !== 0) return byWins;
1306
+ for (const metric of context.input.costMetricOrder) {
1307
+ const byCost = left.incrementalMedianCost[metric] - right.incrementalMedianCost[metric];
1308
+ if (byCost !== 0) return byCost;
1309
+ }
1310
+ const byBytes = summaryById.get(left.additionCandidateId).candidate.artifactBytes - summaryById.get(right.additionCandidateId).candidate.artifactBytes;
1311
+ if (byBytes !== 0) return byBytes;
1312
+ return context.candidateIndex.get(left.additionCandidateId) - context.candidateIndex.get(right.additionCandidateId);
1313
+ }
1314
+ function compareCandidates(context, summaryById, comparatorCandidateId, treatmentCandidateId) {
1315
+ const winsTaskIds = [];
1316
+ const regressionTaskIds = [];
1317
+ const missingTaskIds = [];
1318
+ for (const taskId of context.input.taskOrder) {
1319
+ const comparator = crossSurfaceRowFor(context, comparatorCandidateId, taskId);
1320
+ const treatment = crossSurfaceRowFor(context, treatmentCandidateId, taskId);
1321
+ if (comparator.completeness !== "complete" || treatment.completeness !== "complete") missingTaskIds.push(taskId);
1322
+ else if (treatment.pass && !comparator.pass) winsTaskIds.push(taskId);
1323
+ else if (!treatment.pass && comparator.pass) regressionTaskIds.push(taskId);
1324
+ }
1325
+ return {
1326
+ comparatorCandidateId,
1327
+ treatmentCandidateId,
1328
+ winsTaskIds,
1329
+ regressionTaskIds,
1330
+ missingTaskIds,
1331
+ paired: pairedComparison(context, comparatorCandidateId, treatmentCandidateId),
1332
+ relativeCost: relativeCosts(context, summaryById.get(treatmentCandidateId), summaryById.get(comparatorCandidateId))
1333
+ };
1334
+ }
1335
+ function pairedComparison(context, comparatorCandidateId, treatmentCandidateId) {
1336
+ const rows = [];
1337
+ for (const taskId of context.input.taskOrder) {
1338
+ rows.push(toPairedArmRow(context, crossSurfaceRowFor(context, comparatorCandidateId, taskId)));
1339
+ rows.push(toPairedArmRow(context, crossSurfaceRowFor(context, treatmentCandidateId, taskId)));
1340
+ }
1341
+ return comparePairedArms(rows, {
1342
+ baselineArm: comparatorCandidateId,
1343
+ treatmentArm: treatmentCandidateId,
1344
+ metricNames: ["score", ...context.input.costMetricOrder],
1345
+ bootstrap: {
1346
+ ...context.input.bootstrap,
1347
+ statistic: "mean"
1348
+ }
1349
+ });
1350
+ }
1351
+ function toPairedArmRow(context, row) {
1352
+ const metrics = Object.fromEntries(context.input.costMetricOrder.map((metric) => [metric, row.cost[metric]]));
1353
+ if (row.completeness === "complete") metrics.score = row.score;
1354
+ return {
1355
+ pairKey: row.taskId,
1356
+ arm: row.candidateId,
1357
+ ...row.completeness === "complete" ? { pass: row.pass } : {},
1358
+ metrics
1359
+ };
1360
+ }
1361
+ function summarizePairEvidence(rows, leftComponentId, rightComponentId, field) {
1362
+ const bothTaskIds = [];
1363
+ const leftOnlyTaskIds = [];
1364
+ const rightOnlyTaskIds = [];
1365
+ const neitherTaskIds = [];
1366
+ const unobservedTaskIds = [];
1367
+ for (const row of rows) {
1368
+ const left = evidenceValue(row, leftComponentId, field);
1369
+ const right = evidenceValue(row, rightComponentId, field);
1370
+ if (left === null || right === null) unobservedTaskIds.push(row.taskId);
1371
+ else if (left && right) bothTaskIds.push(row.taskId);
1372
+ else if (left) leftOnlyTaskIds.push(row.taskId);
1373
+ else if (right) rightOnlyTaskIds.push(row.taskId);
1374
+ else neitherTaskIds.push(row.taskId);
1375
+ }
1376
+ return {
1377
+ bothTaskIds,
1378
+ leftOnlyTaskIds,
1379
+ rightOnlyTaskIds,
1380
+ neitherTaskIds,
1381
+ unobservedTaskIds
1382
+ };
1383
+ }
1384
+ function relativeCosts(context, treatment, comparator) {
1385
+ return Object.fromEntries(context.input.costMetricOrder.map((metric) => {
1386
+ const treatmentMedian = treatment.costs[metric].median;
1387
+ const comparatorMedian = comparator.costs[metric].median;
1388
+ return [metric, {
1389
+ treatmentMedian,
1390
+ comparatorMedian,
1391
+ medianDelta: treatmentMedian - comparatorMedian,
1392
+ medianRatio: comparatorMedian === 0 ? treatmentMedian === 0 ? 1 : null : treatmentMedian / comparatorMedian
1393
+ }];
1394
+ }));
1395
+ }
1396
+ function withinCostLimits(context, treatment, baseline) {
1397
+ const relative = relativeCosts(context, treatment, baseline);
1398
+ return Object.entries(context.input.selection.maximumMedianCostRatioToBaseline).every(([metric, limit]) => {
1399
+ const ratio = relative[metric].medianRatio;
1400
+ return ratio !== null && ratio <= limit;
1401
+ });
1402
+ }
1403
+ function distribution(values) {
1404
+ const sorted = [...values].sort((left, right) => left - right);
1405
+ const total = sorted.reduce((sum, value) => sum + value, 0);
1406
+ const middle = Math.floor(sorted.length / 2);
1407
+ const median = sorted.length % 2 === 1 ? sorted[middle] : (sorted[middle - 1] + sorted[middle]) / 2;
1408
+ return {
1409
+ n: sorted.length,
1410
+ min: sorted[0],
1411
+ median,
1412
+ mean: total / sorted.length,
1413
+ max: sorted[sorted.length - 1],
1414
+ total
1415
+ };
1416
+ }
1417
+ function evidenceValue(row, componentId, field) {
1418
+ return row.componentEvidence.find((evidence) => evidence.componentId === componentId)[field];
1419
+ }
1420
+ function meanOrNull$1(values) {
1421
+ return values.length === 0 ? null : values.reduce((sum, value) => sum + value, 0) / values.length;
1422
+ }
1423
+ function unique(values) {
1424
+ return [...new Set(values)];
1425
+ }
1426
+ //#endregion
1427
+ //#region src/campaign/fixtures.ts
1428
+ /** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
1429
+ function discoverEvalFixtures(evalsDir) {
1430
+ const root = resolve(evalsDir);
1431
+ if (!existsSync(root)) throw new Error(`discoverEvalFixtures: evalsDir not found: ${root}`);
1432
+ const fixtures = [];
1433
+ const walk = (dir, base = "") => {
1434
+ for (const entry of readdirSync(dir).sort()) {
1435
+ if (entry.startsWith(".") || entry === "node_modules") continue;
1436
+ const fullPath = join(dir, entry);
1437
+ if (!statSync(fullPath).isDirectory()) continue;
1438
+ const name = base ? `${base}/${entry}` : entry;
1439
+ if (existsWithExactCase(fullPath, "PROMPT.md")) fixtures.push(name);
1440
+ else walk(fullPath, name);
1441
+ }
1442
+ };
1443
+ walk(root);
1444
+ return fixtures;
1445
+ }
1446
+ /**
1447
+ * Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
1448
+ * `vitest` validation) and content-fingerprints the full file set for cache identity.
1449
+ */
1450
+ function loadEvalFixture(evalsDir, name, options = {}) {
1451
+ const validation = options.validation ?? "vitest";
1452
+ const root = resolve(evalsDir);
1453
+ const fixturePath = resolve(root, name);
1454
+ assertInside(root, fixturePath, name);
1455
+ if (!existsSync(fixturePath) || !statSync(fixturePath).isDirectory()) throw new Error(`loadEvalFixture: fixture not found: ${name}`);
1456
+ if (!existsWithExactCase(fixturePath, "PROMPT.md")) throw new Error(`loadEvalFixture: ${name} is missing exact-case PROMPT.md`);
1457
+ const promptPath = join(fixturePath, "PROMPT.md");
1458
+ const evalPath = resolveEvalPath(fixturePath);
1459
+ const packageJsonPath = existsWithExactCase(fixturePath, "package.json") ? join(fixturePath, "package.json") : void 0;
1460
+ if (validation !== "none") {
1461
+ if (!evalPath) throw new Error(`loadEvalFixture: ${name} is missing exact-case EVAL.ts or EVAL.tsx`);
1462
+ if (!packageJsonPath) throw new Error(`loadEvalFixture: ${name} is missing exact-case package.json`);
1463
+ assertModulePackage(packageJsonPath, name);
1464
+ }
1465
+ const files = collectFixtureFiles(fixturePath);
1466
+ return {
1467
+ name,
1468
+ path: fixturePath,
1469
+ promptPath,
1470
+ evalPath,
1471
+ packageJsonPath,
1472
+ prompt: readFileSync(promptPath, "utf8"),
1473
+ files,
1474
+ fingerprint: contentHash({
1475
+ files,
1476
+ config: options.fingerprintConfig ?? null
1477
+ })
1478
+ };
1479
+ }
1480
+ /** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
1481
+ function loadEvalFixtureScenarios(evalsDir, options = {}) {
1482
+ return (options.names ?? discoverEvalFixtures(evalsDir)).map((name) => {
1483
+ const fixture = loadEvalFixture(evalsDir, name, options);
1484
+ return {
1485
+ id: fixture.name,
1486
+ kind: "eval-fixture",
1487
+ tags: ["eval-fixture"],
1488
+ fixtureName: fixture.name,
1489
+ fixturePath: fixture.path,
1490
+ promptPath: fixture.promptPath,
1491
+ evalPath: fixture.evalPath,
1492
+ packageJsonPath: fixture.packageJsonPath,
1493
+ prompt: fixture.prompt,
1494
+ fingerprint: fixture.fingerprint
1495
+ };
1496
+ });
1497
+ }
1498
+ /**
1499
+ * Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
1500
+ * and returns the plan plus each fixture's name/path/fingerprint.
1501
+ */
1502
+ function planEvalFixtureRun(options) {
1503
+ const scenarios = loadEvalFixtureScenarios(options.evalsDir, {
1504
+ names: options.names,
1505
+ validation: options.validation,
1506
+ fingerprintConfig: options.fingerprintConfig
1507
+ });
1508
+ return {
1509
+ ...planCampaignRun({
1510
+ scenarios,
1511
+ dispatchRef: options.dispatchRef ?? "eval-fixture-dispatch",
1512
+ judges: options.judges,
1513
+ seed: options.seed,
1514
+ reps: options.reps,
1515
+ resumable: options.resumable,
1516
+ runDir: options.runDir,
1517
+ storage: options.storage
1518
+ }),
1519
+ fixtures: scenarios.map((scenario) => ({
1520
+ fixtureName: scenario.fixtureName,
1521
+ fixturePath: scenario.fixturePath,
1522
+ fingerprint: scenario.fingerprint
1523
+ }))
1524
+ };
1525
+ }
1526
+ function existsWithExactCase(dirPath, fileName) {
1527
+ try {
1528
+ return readdirSync(dirPath).includes(fileName);
1529
+ } catch {
1530
+ return false;
1531
+ }
1532
+ }
1533
+ function resolveEvalPath(fixturePath) {
1534
+ if (existsWithExactCase(fixturePath, "EVAL.ts")) return join(fixturePath, "EVAL.ts");
1535
+ if (existsWithExactCase(fixturePath, "EVAL.tsx")) return join(fixturePath, "EVAL.tsx");
1536
+ }
1537
+ function assertModulePackage(packageJsonPath, name) {
1538
+ let parsed;
1539
+ try {
1540
+ parsed = JSON.parse(readFileSync(packageJsonPath, "utf8"));
1541
+ } catch (err) {
1542
+ throw new Error(`loadEvalFixture: ${name} package.json is invalid JSON: ${err instanceof Error ? err.message : String(err)}`);
1543
+ }
1544
+ if (typeof parsed !== "object" || parsed === null || parsed.type !== "module") throw new Error(`loadEvalFixture: ${name} package.json must set "type": "module"`);
1545
+ }
1546
+ function collectFixtureFiles(fixturePath, base = "") {
1547
+ const files = [];
1548
+ for (const entry of readdirSync(join(fixturePath, base)).sort()) {
1549
+ if (entry === "node_modules" || entry === ".git") continue;
1550
+ const relativePath = base ? `${base}/${entry}` : entry;
1551
+ const fullPath = join(fixturePath, relativePath);
1552
+ if (statSync(fullPath).isDirectory()) {
1553
+ files.push(...collectFixtureFiles(fixturePath, relativePath));
1554
+ continue;
1555
+ }
1556
+ const bytes = readFileSync(fullPath);
1557
+ files.push({
1558
+ path: relativePath,
1559
+ sha256: createHash("sha256").update(bytes).digest("hex"),
1560
+ bytes: bytes.byteLength
1561
+ });
1562
+ }
1563
+ return files;
1564
+ }
1565
+ function assertInside(root, target, label) {
1566
+ const rel = relative(root, target);
1567
+ if (rel === "" || !rel.startsWith("..") && !isAbsolute(rel)) return;
1568
+ throw new Error(`loadEvalFixture: fixture path escapes evalsDir: ${label}`);
1569
+ }
1570
+ //#endregion
1571
+ //#region src/campaign/gates/neutralization-gate.ts
1572
+ /** Mean of a numeric array; 0 for an empty array (callers guard n separately). */
1573
+ function mean$1(xs) {
1574
+ return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
1575
+ }
1576
+ /** Paired mean held-out lift of `arm` over baseline, on the in-scope cells. */
1577
+ function pairedLift(arm, baseline, scenarioIds) {
1578
+ const paired = pairHoldout(arm, baseline, scenarioIds, (s) => s.composite);
1579
+ const deltas = paired.after.map((a, i) => a - (paired.before[i] ?? 0));
1580
+ return {
1581
+ lift: mean$1(deltas),
1582
+ n: deltas.length
1583
+ };
1584
+ }
1585
+ /**
1586
+ * Composable placebo gate: ships only when the candidate's held-out lift is NOT
1587
+ * mostly reproduced by a footprint-matched neutralized variant.
1588
+ */
1589
+ function neutralizationGate(options) {
1590
+ const maxDecorativeFraction = options.maxDecorativeFraction ?? .5;
1591
+ return {
1592
+ name: "neutralizationGate",
1593
+ async decide(ctx) {
1594
+ if (!ctx.baselineJudgeScores) throw new Error("neutralizationGate: ctx.baselineJudgeScores is required — the placebo control measures lift OVER baseline.");
1595
+ if (!ctx.neutralizedJudgeScores) throw new Error("neutralizationGate: ctx.neutralizedJudgeScores is required. It is populated by runImprovementLoop only when a `neutralize` function is supplied — composing this gate without that wiring would pass an unproven candidate.");
1596
+ const scenarioIds = new Set(options.scenarios.map((s) => s.id));
1597
+ const cand = pairedLift(ctx.judgeScores, ctx.baselineJudgeScores, scenarioIds);
1598
+ const neut = pairedLift(ctx.neutralizedJudgeScores, ctx.baselineJudgeScores, scenarioIds);
1599
+ if (cand.lift <= 0) return {
1600
+ decision: "hold",
1601
+ reasons: [`neutralization: candidate held-out lift ${cand.lift.toFixed(3)} ≤ 0 — no positive lift to attribute to content`],
1602
+ contributingGates: [{
1603
+ name: "neutralizationGate",
1604
+ status: "fail",
1605
+ detail: {
1606
+ candidateLift: cand.lift,
1607
+ neutralizedLift: neut.lift,
1608
+ n: cand.n
1609
+ }
1610
+ }],
1611
+ delta: cand.lift
1612
+ };
1613
+ const decorativeFraction = neut.lift / cand.lift;
1614
+ const passed = decorativeFraction < maxDecorativeFraction;
1615
+ const pct = (decorativeFraction * 100).toFixed(0);
1616
+ return {
1617
+ decision: passed ? "ship" : "hold",
1618
+ reasons: passed ? [`neutralization: content is causal — blanked variant reproduces ${pct}% of the lift (< ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate Δ ${cand.lift.toFixed(3)}, neutralized Δ ${neut.lift.toFixed(3)}`] : [`neutralization: lift is DECORATIVE — blanking the content (footprint-matched) reproduces ${pct}% of the lift (≥ ${(maxDecorativeFraction * 100).toFixed(0)}%); candidate Δ ${cand.lift.toFixed(3)}, neutralized Δ ${neut.lift.toFixed(3)}`],
1619
+ contributingGates: [{
1620
+ name: "neutralizationGate",
1621
+ status: passed ? "pass" : "fail",
1622
+ detail: {
1623
+ candidateLift: cand.lift,
1624
+ neutralizedLift: neut.lift,
1625
+ decorativeFraction,
1626
+ maxDecorativeFraction,
1627
+ n: cand.n
1628
+ }
1629
+ }],
1630
+ delta: cand.lift
1631
+ };
1632
+ }
1633
+ };
1634
+ }
1635
+ //#endregion
1636
+ //#region src/campaign/gates/sequential.ts
1637
+ /**
1638
+ * Anytime-valid sequential promotion gate — an e-process (betting
1639
+ * test-martingale, see `eProcess` in `statistics.ts`) over paired
1640
+ * per-scenario deltas, so a campaign stops the MOMENT evidence decides
1641
+ * instead of burning a fixed-n budget. Decisions remain valid at any
1642
+ * data-dependent stopping time (Ville's inequality), which is exactly what
1643
+ * fixed-n machinery cannot offer: peeking at a bootstrap CI after every
1644
+ * observation and stopping on the first significant peek inflates type-I
1645
+ * error far beyond alpha.
1646
+ *
1647
+ * REPLACES, never layers on, a fixed-n gate. Running `heldoutSignificance`
1648
+ * or `paretoSignificanceGate` repeatedly on a growing sample and stopping
1649
+ * early is optional stopping no matter how it is dressed up; this gate is
1650
+ * the valid way to stop early. Use one or the other per evidence stream.
1651
+ *
1652
+ * Pre-registration binding: anytime validity holds only for the
1653
+ * PRE-REGISTERED statistic. When a `SignedManifest` is bound, the gate takes
1654
+ * alpha from `manifest.alpha`, the observation budget from
1655
+ * `manifest.preRegisteredN`, orients deltas by `manifest.direction`, and
1656
+ * shifts the null boundary by `manifest.minEffect` — re-deciding the same
1657
+ * stream under different parameters after seeing data would reopen optional
1658
+ * stopping under a fancier name. The manifest's content hash is verified at
1659
+ * construction (sync, same `sha256-content` scheme as `signManifest`).
1660
+ *
1661
+ * Non-iid caveat (stated honestly): the supermartingale guarantee needs each
1662
+ * delta's conditional mean under H0 to stay ≤ the null boundary given the
1663
+ * past — exchangeable scenario deltas suffice. Scenario streams ordered by
1664
+ * difficulty or by scenario family violate this; `decide(ctx)` therefore
1665
+ * shuffles the paired deltas with a SEEDED permutation by default (the
1666
+ * permutation is data-independent, so bet predictability is preserved).
1667
+ * Stratified betting (per-stratum λ) is future work, not implemented here.
1668
+ */
1669
+ /** Sync twin of `verifyManifest` — same `sha256-content` scheme
1670
+ * (sha256 over the canonicalized manifest minus contentHash/algo), via
1671
+ * node:crypto so gate construction can stay synchronous and fail loud
1672
+ * before any observation is consumed. */
1673
+ function verifyManifestSync(m) {
1674
+ if (m.algo !== void 0 && m.algo !== "sha256-content") throw new Error(`sequentialPairedGate: unrecognized manifest hash algo '${m.algo}'`);
1675
+ const { contentHash, algo: _algo, ...rest } = m;
1676
+ const bytes = JSON.stringify(canonicalize(rest));
1677
+ return createHash("sha256").update(bytes, "utf8").digest("hex") === contentHash;
1678
+ }
1679
+ function resolveConfig(opts) {
1680
+ const m = opts.preRegistration;
1681
+ let alpha;
1682
+ let maxN;
1683
+ let direction;
1684
+ let minEffect;
1685
+ if (m) {
1686
+ if (!verifyManifestSync(m)) throw new Error(`sequentialPairedGate: pre-registration manifest '${m.id}' content hash mismatch (tampered)`);
1687
+ if (opts.alpha !== void 0 && opts.alpha !== m.alpha) throw new Error(`sequentialPairedGate: alpha ${opts.alpha} conflicts with pre-registered alpha ${m.alpha} — the registered statistic is the only one anytime validity covers`);
1688
+ if (opts.maxN !== void 0 && opts.maxN !== m.preRegisteredN) throw new Error(`sequentialPairedGate: maxN ${opts.maxN} conflicts with pre-registered N ${m.preRegisteredN}`);
1689
+ alpha = m.alpha;
1690
+ maxN = m.preRegisteredN;
1691
+ direction = m.direction;
1692
+ minEffect = m.minEffect;
1693
+ } else {
1694
+ alpha = opts.alpha ?? .05;
1695
+ if (opts.maxN === void 0) throw new Error("sequentialPairedGate: maxN is required (or bind a preRegistration manifest whose preRegisteredN is the budget) — an unbounded stream has no pre-registered budget");
1696
+ maxN = opts.maxN;
1697
+ direction = "increase";
1698
+ minEffect = 0;
1699
+ }
1700
+ const scale = opts.scale ?? 1;
1701
+ if (!Number.isFinite(scale) || scale <= 0) throw new Error(`sequentialPairedGate: scale must be > 0, got ${scale}`);
1702
+ if (!Number.isInteger(maxN) || maxN < 1) throw new Error(`sequentialPairedGate: maxN must be a positive integer, got ${maxN}`);
1703
+ const minN = opts.minN ?? 5;
1704
+ if (!Number.isInteger(minN) || minN < 1 || minN > maxN) throw new Error(`sequentialPairedGate: minN must be an integer in [1, maxN=${maxN}], got ${minN}`);
1705
+ if (!Number.isFinite(minEffect) || minEffect < 0 || minEffect >= scale) throw new Error(`sequentialPairedGate: minEffect must be in [0, scale=${scale}), got ${minEffect}`);
1706
+ const nullMean = .5 + minEffect / (2 * scale);
1707
+ return {
1708
+ alpha,
1709
+ minN,
1710
+ maxN,
1711
+ maxBet: opts.maxBet ?? .5,
1712
+ scale,
1713
+ shuffleSeed: opts.shuffleSeed ?? 1337,
1714
+ direction,
1715
+ nullMean,
1716
+ minEffect
1717
+ };
1718
+ }
1719
+ function makeStream(cfg) {
1720
+ const proc = eProcess({
1721
+ alpha: cfg.alpha,
1722
+ maxBet: cfg.maxBet,
1723
+ nullMean: cfg.nullMean
1724
+ });
1725
+ const threshold = 1 / cfg.alpha;
1726
+ let terminal;
1727
+ return {
1728
+ observe(delta) {
1729
+ if (terminal === "undecided-at-maxN") throw new Error(`sequentialPairedGate: pre-registered maxN=${cfg.maxN} exhausted — extending the stream after seeing the result reopens optional stopping; start a NEW pre-registered test`);
1730
+ if (!Number.isFinite(delta) || Math.abs(delta) > cfg.scale) throw new Error(`sequentialPairedGate: delta ${delta} outside ±scale=${cfg.scale} — pass the judge's native scale explicitly (detectScale helps pick 1 vs 100)`);
1731
+ const d = cfg.direction === "decrease" ? -delta : delta;
1732
+ const step = proc.update((d / cfg.scale + 1) / 2);
1733
+ if (terminal === void 0 && step.n >= cfg.minN && step.wealth >= threshold) terminal = "promote";
1734
+ if (terminal === "promote") return {
1735
+ decision: "promote",
1736
+ eValue: step.wealth,
1737
+ n: step.n,
1738
+ reason: `e-value ${step.wealth.toFixed(2)} ≥ 1/α=${threshold.toFixed(2)} at n=${step.n} (minN=${cfg.minN}): the paired improvement exceeds ${cfg.minEffect} at anytime-valid level α=${cfg.alpha}`
1739
+ };
1740
+ if (step.n >= cfg.maxN) {
1741
+ terminal = "undecided-at-maxN";
1742
+ return {
1743
+ decision: "undecided-at-maxN",
1744
+ eValue: step.wealth,
1745
+ n: step.n,
1746
+ reason: `undecided at pre-registered maxN=${cfg.maxN} (e-value ${step.wealth.toFixed(2)} < 1/α=${threshold.toFixed(2)}). This is NOT evidence of no effect — the effect may be real but smaller than this budget can detect; re-register with a larger N to test that`
1747
+ };
1748
+ }
1749
+ return {
1750
+ decision: "continue",
1751
+ eValue: step.wealth,
1752
+ n: step.n,
1753
+ reason: `e-value ${step.wealth.toFixed(2)} < 1/α=${threshold.toFixed(2)} at n=${step.n}/${cfg.maxN} — keep observing`
1754
+ };
1755
+ },
1756
+ state() {
1757
+ return {
1758
+ ...proc.state(),
1759
+ decision: terminal ?? "continue"
1760
+ };
1761
+ }
1762
+ };
1763
+ }
1764
+ /** Data-independent in-place Fisher–Yates with a seeded PRNG — the permutation
1765
+ * depends only on the seed, never the values, so bet predictability survives. */
1766
+ function seededShuffle(items, seed) {
1767
+ const rng = mulberry32(seed);
1768
+ for (let i = items.length - 1; i > 0; i--) {
1769
+ const j = Math.floor(rng() * (i + 1));
1770
+ const tmp = items[i];
1771
+ items[i] = items[j];
1772
+ items[j] = tmp;
1773
+ }
1774
+ return items;
1775
+ }
1776
+ /**
1777
+ * Anytime-valid sequential paired gate. Conforms to the existing `Gate`
1778
+ * contract (`decide(ctx)` consumes candidate vs baseline judge scores via
1779
+ * `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
1780
+ * never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
1781
+ * that score cells incrementally and want to stop mid-stream.
1782
+ *
1783
+ * Decision mapping onto the substrate's five-valued `GateDecision`:
1784
+ * - 'promote' → 'ship'
1785
+ * - 'continue' → 'need_more_work' (stream ended before maxN with
1786
+ * the e-value undecided — more reps could decide)
1787
+ * - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
1788
+ * evidence of no effect (never a silent default)
1789
+ */
1790
+ function sequentialPairedGate(options) {
1791
+ const cfg = resolveConfig(options);
1792
+ const name = options.name ?? "sequentialPairedGate";
1793
+ const manifest = options.preRegistration;
1794
+ const observeStream = makeStream(cfg);
1795
+ return {
1796
+ name,
1797
+ observe(delta) {
1798
+ return observeStream.observe(delta);
1799
+ },
1800
+ state() {
1801
+ return observeStream.state();
1802
+ },
1803
+ async decide(ctx) {
1804
+ if (!ctx.baselineJudgeScores) throw new Error(`${name}: ctx.baselineJudgeScores is required — falling back to the candidate's own scores would compare the candidate against itself (delta 0, silent no-op)`);
1805
+ const scenarioIds = new Set(ctx.scenarios.map((s) => s.id));
1806
+ const paired = pairHoldout(ctx.judgeScores, ctx.baselineJudgeScores, scenarioIds, (s) => s.composite);
1807
+ const deltas = paired.after.map((a, i) => a - paired.before[i]);
1808
+ seededShuffle(deltas, cfg.shuffleSeed);
1809
+ const stream = makeStream(cfg);
1810
+ let last;
1811
+ for (const d of deltas) {
1812
+ last = stream.observe(d);
1813
+ if (last.decision !== "continue") break;
1814
+ }
1815
+ const detail = {
1816
+ ...stream.state(),
1817
+ minN: cfg.minN,
1818
+ maxN: cfg.maxN,
1819
+ scale: cfg.scale,
1820
+ shuffleSeed: cfg.shuffleSeed,
1821
+ direction: cfg.direction,
1822
+ minEffect: cfg.minEffect,
1823
+ pairedN: deltas.length,
1824
+ ...manifest ? {
1825
+ preRegisteredId: manifest.id,
1826
+ metric: manifest.metric
1827
+ } : {}
1828
+ };
1829
+ const meanDelta = deltas.length === 0 ? void 0 : deltas.reduce((s, d) => s + d, 0) / deltas.length;
1830
+ if (last === void 0) return {
1831
+ decision: "need_more_work",
1832
+ reasons: [`${name}: no paired holdout observations — nothing to test`],
1833
+ contributingGates: [{
1834
+ name,
1835
+ status: "not_evaluated",
1836
+ detail
1837
+ }]
1838
+ };
1839
+ const decision = last.decision === "promote" ? "ship" : last.decision === "continue" ? "need_more_work" : "hold";
1840
+ return {
1841
+ decision,
1842
+ reasons: [`${name}: ${last.reason}`],
1843
+ contributingGates: [{
1844
+ name,
1845
+ status: decision === "ship" ? "pass" : decision === "need_more_work" ? "not_evaluated" : "fail",
1846
+ detail
1847
+ }],
1848
+ delta: meanDelta
1849
+ };
1850
+ }
1851
+ };
1852
+ }
1853
+ /**
1854
+ * `SurfaceProposer.decide` adapter — stops the optimization loop the moment
1855
+ * the e-process decides the loop has produced a real improvement, instead of
1856
+ * always running `maxGenerations`.
1857
+ *
1858
+ * Stream: for each generation g ≥ 1, the per-scenario composite deltas of
1859
+ * generation g's top candidate vs the generation-0 top candidate (the
1860
+ * incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
1861
+ * surface improves any scenario's expected composite over the incumbent —
1862
+ * under it every delta has conditional mean ≤ 0 and the e-process is valid.
1863
+ * Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
1864
+ * gate (which re-scores on HELD-OUT data — this adapter only spends the
1865
+ * exploration budget, it never promotes).
1866
+ *
1867
+ * Honesty caveats: (1) the incumbent's scores are measured once and shared
1868
+ * across all generations' deltas, so type-I control is exact only insofar as
1869
+ * those scores approximate the incumbent's true per-scenario means (more reps
1870
+ * → tighter); (2) an UNDECIDED process never stops the loop — absence of a
1871
+ * crossing is NOT evidence of no effect, so the loop simply runs its normal
1872
+ * course. Calling the adapter repeatedly with a growing history consumes each
1873
+ * generation exactly once (re-feeding an already-seen record would double-count
1874
+ * evidence).
1875
+ */
1876
+ function sequentialDecide(options = {}) {
1877
+ const alpha = options.alpha ?? .05;
1878
+ const minN = options.minN ?? 5;
1879
+ const scale = options.scale ?? 1;
1880
+ if (!Number.isFinite(scale) || scale <= 0) throw new Error(`sequentialDecide: scale must be > 0, got ${scale}`);
1881
+ const proc = eProcess({
1882
+ alpha,
1883
+ maxBet: options.maxBet ?? .5,
1884
+ nullMean: .5
1885
+ });
1886
+ const threshold = 1 / alpha;
1887
+ let processedGenerations = 0;
1888
+ let reference;
1889
+ let stopped;
1890
+ const topCandidate = (record) => {
1891
+ if (record.candidates.length === 0) throw new Error(`sequentialDecide: generation ${record.generationIndex} has no candidates — cannot extract a top candidate`);
1892
+ return record.candidates.reduce((best, candidate) => {
1893
+ if (best.composite === null) return candidate;
1894
+ if (candidate.composite === null) return best;
1895
+ return candidate.composite > best.composite ? candidate : best;
1896
+ });
1897
+ };
1898
+ const decide = ({ history }) => {
1899
+ if (stopped) return stopped;
1900
+ if (history.length === 0) return { stop: false };
1901
+ if (reference === void 0) reference = new Map(topCandidate(history[0]).scenarios.map((s) => [s.scenarioId, s.composite]));
1902
+ for (let g = Math.max(1, processedGenerations); g < history.length; g++) {
1903
+ const top = topCandidate(history[g]);
1904
+ const byScenario = new Map(top.scenarios.map((s) => [s.scenarioId, s.composite]));
1905
+ for (const [scenarioId, refComposite] of reference) {
1906
+ const candComposite = byScenario.get(scenarioId);
1907
+ if (candComposite === void 0) throw new Error(`sequentialDecide: generation ${history[g].generationIndex} top candidate is missing scenario '${scenarioId}' — generations must score the same scenario set to pair`);
1908
+ const delta = candComposite - refComposite;
1909
+ if (!Number.isFinite(delta) || Math.abs(delta) > scale) throw new Error(`sequentialDecide: paired delta ${delta} outside ±scale=${scale} on scenario '${scenarioId}' — pass the composite scale explicitly`);
1910
+ const step = proc.update((delta / scale + 1) / 2);
1911
+ if (step.n >= minN && step.wealth >= threshold) {
1912
+ stopped = {
1913
+ stop: true,
1914
+ reason: `sequential e-process decided at generation ${history[g].generationIndex}: e-value ${step.wealth.toFixed(2)} ≥ 1/α=${threshold.toFixed(2)} after n=${step.n} paired deltas vs the generation-0 incumbent — the improvement is real at α=${alpha}; stop exploring and promote via the gate`
1915
+ };
1916
+ processedGenerations = history.length;
1917
+ return stopped;
1918
+ }
1919
+ }
1920
+ }
1921
+ processedGenerations = history.length;
1922
+ return { stop: false };
1923
+ };
1924
+ decide.state = () => proc.state();
1925
+ return decide;
1926
+ }
1927
+ //#endregion
1928
+ //#region src/campaign/grounded-reflection.ts
1929
+ /**
1930
+ * Deterministic per-field diff of call arguments between passing and failing
1931
+ * rollouts. A field set by failing rollouts but left unset by passing ones is
1932
+ * the classic poison-input signature; a field whose values differ across the
1933
+ * split points at the correct value. Feed `text` to the reviser verbatim.
1934
+ */
1935
+ function rolloutArgumentDiff(rollouts, opts = {}) {
1936
+ const passThreshold = opts.passThreshold ?? 1;
1937
+ const maxValues = opts.maxValuesPerField ?? 4;
1938
+ const collect = (pass) => {
1939
+ const byField = /* @__PURE__ */ new Map();
1940
+ for (const r of rollouts) {
1941
+ if (pass !== r.score >= passThreshold) continue;
1942
+ for (const c of r.calls) for (const [k, v] of Object.entries(c.args)) {
1943
+ const set = byField.get(k) ?? /* @__PURE__ */ new Set();
1944
+ set.add(String(v));
1945
+ byField.set(k, set);
1946
+ }
1947
+ }
1948
+ return byField;
1949
+ };
1950
+ const passing = collect(true);
1951
+ const failing = collect(false);
1952
+ const lines = [];
1953
+ for (const field of /* @__PURE__ */ new Set([...passing.keys(), ...failing.keys()])) {
1954
+ const pv = [...passing.get(field) ?? []].slice(0, maxValues);
1955
+ const fv = [...failing.get(field) ?? []].slice(0, maxValues);
1956
+ const render = (vals) => vals.length ? JSON.stringify(vals) : "NOT SET (omitted)";
1957
+ lines.push(` ${field}: passing runs -> ${render(pv)} | failing runs -> ${render(fv)}`);
1958
+ }
1959
+ const lower = (m) => {
1960
+ const out = /* @__PURE__ */ new Set();
1961
+ for (const vals of m.values()) for (const v of vals) out.add(v.toLowerCase());
1962
+ return out;
1963
+ };
1964
+ return {
1965
+ text: lines.join("\n") || " (no calls observed)",
1966
+ passingValues: lower(passing),
1967
+ failingValues: lower(failing)
1968
+ };
1969
+ }
1970
+ /**
1971
+ * Scan revised artifact text for single-quoted single-word literals (the
1972
+ * "use exactly 'new'" pattern) that appear in no passing rollout's argument
1973
+ * values. Multi-word quotes pass (they are prose, not prescriptions).
1974
+ * Callers should reject on `harmful` (with a bounded retry) and at most log
1975
+ * `ungrounded` - see the module header for why the severities differ.
1976
+ */
1977
+ function classifyUngroundedLiterals(text, diff) {
1978
+ const ungrounded = /* @__PURE__ */ new Set();
1979
+ for (const m of text.matchAll(/'([a-z][a-z_-]{1,19})'/gi)) {
1980
+ const w = m[1].toLowerCase();
1981
+ if (!diff.passingValues.has(w)) ungrounded.add(w);
1982
+ }
1983
+ const all = [...ungrounded];
1984
+ return {
1985
+ ungrounded: all,
1986
+ harmful: all.filter((w) => diff.failingValues.has(w))
1987
+ };
1988
+ }
1989
+ //#endregion
1990
+ //#region src/campaign/labeled-store/fs-adapter.ts
1991
+ /**
1992
+ * Filesystem `LabeledScenarioStore` adapter. The default capture sink for
1993
+ * traces + eval artifacts. Production deployments typically swap for a
1994
+ * Turso/SQLite adapter (same interface).
1995
+ *
1996
+ * Records land as one JSONL file per source under `<root>/<source>.jsonl`.
1997
+ * Each line is a `LabeledScenarioRecord`. Append-only — no in-place edits.
1998
+ *
1999
+ * Safety properties enforced at write-time:
2000
+ *
2001
+ * - **Provenance required**: writes without `source`, `sourceVersionHash`,
2002
+ * `capturedAt`, `redactionStatus` are rejected. Closes the alignment
2003
+ * reviewer's data-poisoning gap.
2004
+ * - **Per-source rate limits**: optional `rateLimitBucket` + `maxWritesPerMinute`
2005
+ * stops a single tenant/source from flooding the store.
2006
+ *
2007
+ * Safety properties enforced at sample-time:
2008
+ *
2009
+ * - **Required split + capturedBefore**: substrate refuses to sample without
2010
+ * an explicit `split` ('train' | 'test') AND a temporal cutoff. Eliminates
2011
+ * accidental train/test contamination.
2012
+ * - **Default training-source filter**: when the store is sampled with
2013
+ * `split: 'train'`, production-trace records are EXCLUDED unless the
2014
+ * caller passes `filter.source: 'production-trace'` explicitly. Closes
2015
+ * the contamination-by-default gap flagged by the senior eval engineer.
2016
+ */
2017
+ /** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
2018
+ var LabeledScenarioStoreError = class extends Error {
2019
+ code;
2020
+ constructor(code, message) {
2021
+ super(message);
2022
+ this.code = code;
2023
+ this.name = "LabeledScenarioStoreError";
2024
+ }
2025
+ };
2026
+ /**
2027
+ * Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
2028
+ * rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
2029
+ */
2030
+ var FsLabeledScenarioStore = class {
2031
+ options;
2032
+ now;
2033
+ rateLimits = /* @__PURE__ */ new Map();
2034
+ constructor(options) {
2035
+ this.options = options;
2036
+ if (!existsSync(options.root)) mkdirSync(options.root, { recursive: true });
2037
+ this.now = options.now ?? Date.now;
2038
+ }
2039
+ async observe(write) {
2040
+ this.assertProvenance(write);
2041
+ this.assertRateLimit(write);
2042
+ const record = this.toRecord(write);
2043
+ appendLine(this.pathForSource(write.source), `${JSON.stringify(record)}\n`);
2044
+ }
2045
+ async sample(args) {
2046
+ if (!args.split) throw new LabeledScenarioStoreError("split_required", "sample() requires an explicit `split` (train | test) — substrate refuses ambiguous reads");
2047
+ if (!args.capturedBefore) throw new LabeledScenarioStoreError("capturedBefore_required", "sample() requires an explicit `capturedBefore` timestamp for temporal-split discipline");
2048
+ const all = [];
2049
+ for (const source of ALL_SOURCES) {
2050
+ if (args.split === "train" && source === "production-trace") {
2051
+ if (!sourceFilterContains(args.filter?.source, "production-trace")) continue;
2052
+ }
2053
+ const path = this.pathForSource(source);
2054
+ if (!existsSync(path)) continue;
2055
+ const lines = readFileSync(path, "utf8").split("\n").filter(Boolean);
2056
+ for (const line of lines) {
2057
+ let record;
2058
+ try {
2059
+ record = JSON.parse(line);
2060
+ } catch {
2061
+ continue;
2062
+ }
2063
+ if (!matchesFilter(record, args, source)) continue;
2064
+ all.push(record);
2065
+ }
2066
+ }
2067
+ all.sort((a, b) => {
2068
+ if (a.capturedAt !== b.capturedAt) return a.capturedAt.localeCompare(b.capturedAt);
2069
+ return a.recordHash.localeCompare(b.recordHash);
2070
+ });
2071
+ return all.slice(0, args.count);
2072
+ }
2073
+ async size() {
2074
+ const bySource = {};
2075
+ const byTrust = {
2076
+ unverified: 0,
2077
+ "verified-signal": 0,
2078
+ "human-rated": 0
2079
+ };
2080
+ let total = 0;
2081
+ for (const source of ALL_SOURCES) {
2082
+ const path = this.pathForSource(source);
2083
+ if (!existsSync(path)) {
2084
+ bySource[source] = 0;
2085
+ continue;
2086
+ }
2087
+ const lines = readFileSync(path, "utf8").split("\n").filter(Boolean);
2088
+ bySource[source] = lines.length;
2089
+ total += lines.length;
2090
+ for (const line of lines) {
2091
+ let trust = "unverified";
2092
+ try {
2093
+ trust = JSON.parse(line).labelTrust ?? "unverified";
2094
+ } catch {}
2095
+ byTrust[trust] += 1;
2096
+ }
2097
+ }
2098
+ return {
2099
+ train: total,
2100
+ test: total,
2101
+ bySource,
2102
+ byTrust
2103
+ };
2104
+ }
2105
+ assertProvenance(write) {
2106
+ if (!write.source) throw new LabeledScenarioStoreError("missing_source", "LabeledScenarioWrite requires `source`");
2107
+ if (!write.sourceVersionHash || write.sourceVersionHash.length === 0) throw new LabeledScenarioStoreError("missing_source_version", "LabeledScenarioWrite requires `sourceVersionHash` (git sha or substrate version)");
2108
+ if (!write.capturedAt) throw new LabeledScenarioStoreError("missing_captured_at", "LabeledScenarioWrite requires `capturedAt` ISO timestamp");
2109
+ if (!write.redactionStatus) throw new LabeledScenarioStoreError("missing_redaction_status", "LabeledScenarioWrite requires explicit `redactionStatus` — raw / redacted-pii / redacted-secrets / fully-redacted");
2110
+ if (!ALL_SOURCES.includes(write.source)) throw new LabeledScenarioStoreError("unknown_source", `LabeledScenarioWrite.source must be one of: ${ALL_SOURCES.join(", ")}`);
2111
+ }
2112
+ assertRateLimit(write) {
2113
+ const cap = this.options.maxWritesPerMinutePerBucket;
2114
+ if (!cap || !write.rateLimitBucket) return;
2115
+ const now = this.now();
2116
+ const windowMs = 6e4;
2117
+ let state = this.rateLimits.get(write.rateLimitBucket);
2118
+ if (!state || now - state.windowStartMs >= windowMs) {
2119
+ state = {
2120
+ bucket: write.rateLimitBucket,
2121
+ windowStartMs: now,
2122
+ count: 0
2123
+ };
2124
+ this.rateLimits.set(write.rateLimitBucket, state);
2125
+ }
2126
+ if (state.count >= cap) throw new LabeledScenarioStoreError("rate_limit_exceeded", `LabeledScenarioStore: bucket ${write.rateLimitBucket} exceeded ${cap} writes/min`);
2127
+ state.count += 1;
2128
+ }
2129
+ toRecord(write) {
2130
+ const recordHash = sha256$1(JSON.stringify({
2131
+ id: write.scenario.id,
2132
+ src: write.source,
2133
+ at: write.capturedAt,
2134
+ ver: write.sourceVersionHash
2135
+ }));
2136
+ return {
2137
+ ...write,
2138
+ recordHash,
2139
+ split: "train"
2140
+ };
2141
+ }
2142
+ pathForSource(source) {
2143
+ return join(this.options.root, `${source}.jsonl`);
2144
+ }
2145
+ };
2146
+ const ALL_SOURCES = [
2147
+ "production-trace",
2148
+ "eval-run",
2149
+ "manual",
2150
+ "red-team",
2151
+ "synthetic"
2152
+ ];
2153
+ function sourceFilterContains(filter, needle) {
2154
+ if (!filter) return false;
2155
+ if (Array.isArray(filter)) return filter.includes(needle);
2156
+ return filter === needle;
2157
+ }
2158
+ function matchesFilter(record, args, source) {
2159
+ if (args.split === "train" && record.capturedAt >= args.capturedBefore) return false;
2160
+ if (args.split === "test" && record.capturedAt < args.capturedBefore) return false;
2161
+ const f = args.filter;
2162
+ if (!f) return true;
2163
+ if (f.kind && record.scenario.kind !== f.kind) return false;
2164
+ if (f.source) {
2165
+ if (!(Array.isArray(f.source) ? f.source : [f.source]).includes(source)) return false;
2166
+ }
2167
+ if (f.minComposite !== void 0 || f.maxComposite !== void 0) {
2168
+ const composites = Object.values(record.judgeScores).map((s) => s.composite);
2169
+ const max = composites.length === 0 ? 0 : Math.max(...composites);
2170
+ if (f.minComposite !== void 0 && max < f.minComposite) return false;
2171
+ if (f.maxComposite !== void 0 && max > f.maxComposite) return false;
2172
+ }
2173
+ if (f.minTrust !== void 0 && labelTrustRank(record.labelTrust) < labelTrustRank(f.minTrust)) return false;
2174
+ return true;
2175
+ }
2176
+ function sha256$1(input) {
2177
+ return createHash("sha256").update(input).digest("hex").slice(0, 16);
2178
+ }
2179
+ function appendLine(path, line) {
2180
+ if (existsSync(path)) writeFileSync(path, readFileSync(path, "utf8") + line);
2181
+ else writeFileSync(path, line);
2182
+ }
2183
+ //#endregion
2184
+ //#region src/campaign/neutralize.ts
2185
+ /**
2186
+ * @module
2187
+ * Footprint-matched neutralization — the placebo control for content-vs-footprint
2188
+ * attribution in a promotion gate.
2189
+ *
2190
+ * A promoted surface can raise a held-out score two different ways:
2191
+ * 1. its CONTENT is informative (the thing we want to promote), or
2192
+ * 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
2193
+ * more authoritative-looking prompt — that the model spends attention on
2194
+ * regardless of what the bytes say.
2195
+ *
2196
+ * A held-out gate proves the candidate beat baseline; it cannot separate (1) from
2197
+ * (2). `neutralizeText` produces a variant that keeps the input's layout and
2198
+ * length while carrying ZERO information, so scoring it isolates the footprint
2199
+ * contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
2200
+ * any lift it still holds over baseline is decorative, and a candidate whose lift
2201
+ * survives neutralization is rejected however large its raw lift.
2202
+ */
2203
+ /** Filler for blanked content. A single ASCII byte, so a run of it preserves an
2204
+ * ASCII source's exact byte length; for multibyte sources it preserves CHARACTER
2205
+ * count and layout (what the tokenizer footprint tracks), not raw byte count. */
2206
+ const FILLER = "#";
2207
+ /**
2208
+ * Blank every non-whitespace character to a 1-byte filler while preserving all
2209
+ * whitespace. Line count, indentation, and word/line lengths are unchanged — so
2210
+ * the neutralized variant has the same layout and (for ASCII) the same byte
2211
+ * footprint as the input, but no readable content. Whitespace is preserved
2212
+ * deliberately: collapsing it would change the token structure and stop the
2213
+ * variant from being a true footprint match.
2214
+ */
2215
+ function neutralizeText(content) {
2216
+ return content.replace(/\S/g, FILLER);
2217
+ }
2218
+ //#endregion
2219
+ //#region src/campaign/presets/playback.ts
2220
+ /**
2221
+ * Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
2222
+ * matrix scores is the `ProducedState` extracted from the driver's event
2223
+ * stream — grade it with `scoreUserStory` (or a judge wrapping it).
2224
+ */
2225
+ function makePlaybackDispatch(driver) {
2226
+ return async (profile, scenario, ctx) => {
2227
+ return extractProducedState(await driver.run(scenario, {
2228
+ ...ctx,
2229
+ profile
2230
+ }));
2231
+ };
2232
+ }
2233
+ /**
2234
+ * Score one story's produced state against its requirements. Thin wrapper over
2235
+ * `verifyCompletion` that builds the gold from the story and returns a
2236
+ * per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
2237
+ * deterministic stub in tests, `createLlmCorrectnessChecker` in production.
2238
+ */
2239
+ async function scoreUserStory(story, state, checkCorrectness) {
2240
+ return {
2241
+ ...await verifyCompletion({
2242
+ taskId: story.id,
2243
+ requirements: story.requirements
2244
+ }, state, checkCorrectness),
2245
+ title: story.title
2246
+ };
2247
+ }
2248
+ /**
2249
+ * Flatten story verdicts into the per-requirement scoreboard — the literal
2250
+ * Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
2251
+ * evidence behind the verdict.
2252
+ */
2253
+ function userStoryScoreboard(verdicts) {
2254
+ const rows = [];
2255
+ for (const v of verdicts) for (const r of v.requirements) rows.push({
2256
+ storyId: v.taskId,
2257
+ storyTitle: v.title,
2258
+ reqId: r.reqId,
2259
+ reqTitle: r.title,
2260
+ status: r.satisfied ? "PASS" : "FAIL",
2261
+ evidence: r.evidence
2262
+ });
2263
+ return rows;
2264
+ }
2265
+ /** Roll the per-requirement rows up into the launch headline counts. */
2266
+ function scoreboardSummary(rows) {
2267
+ const byStory = /* @__PURE__ */ new Map();
2268
+ let passed = 0;
2269
+ for (const r of rows) {
2270
+ const s = byStory.get(r.storyId) ?? {
2271
+ total: 0,
2272
+ passed: 0
2273
+ };
2274
+ s.total++;
2275
+ if (r.status === "PASS") {
2276
+ s.passed++;
2277
+ passed++;
2278
+ }
2279
+ byStory.set(r.storyId, s);
2280
+ }
2281
+ let storiesFullyComplete = 0;
2282
+ for (const s of byStory.values()) if (s.total > 0 && s.passed === s.total) storiesFullyComplete++;
2283
+ return {
2284
+ stories: byStory.size,
2285
+ storiesFullyComplete,
2286
+ requirements: rows.length,
2287
+ passed,
2288
+ failed: rows.length - passed,
2289
+ passRate: rows.length === 0 ? 0 : passed / rows.length
2290
+ };
2291
+ }
2292
+ function escapeCell(s) {
2293
+ return s.replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
2294
+ }
2295
+ function truncate(s, max) {
2296
+ return s.length <= max ? s : `${s.slice(0, Math.max(0, max - 1))}…`;
2297
+ }
2298
+ /**
2299
+ * Render the scoreboard as a launch-readiness Markdown document — the literal
2300
+ * "tick off every user story" artifact: a headline roll-up, the open tickets
2301
+ * (FAIL rows) up top as the launch blockers, then a per-story table of
2302
+ * requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
2303
+ * rows in, same bytes out (no clock/random), so it is safe to snapshot.
2304
+ */
2305
+ function renderScoreboardMarkdown(rows, opts = {}) {
2306
+ const maxEv = opts.maxEvidenceChars ?? 160;
2307
+ const sum = scoreboardSummary(rows);
2308
+ const pct = (n) => `${Math.round(n * 100)}%`;
2309
+ const ev = (e) => escapeCell(truncate(e.join("; "), maxEv)) || "—";
2310
+ const out = [`# ${opts.title ?? "Product-flow playback scoreboard"}`, ""];
2311
+ if (opts.meta) {
2312
+ for (const [k, v] of Object.entries(opts.meta)) out.push(`- **${k}:** ${v}`);
2313
+ out.push("");
2314
+ }
2315
+ out.push(`**${sum.storiesFullyComplete}/${sum.stories}** user stories fully shipped · **${sum.passed}/${sum.requirements}** requirements passing (${pct(sum.passRate)}) · **${sum.failed}** open`, "");
2316
+ const fails = rows.filter((r) => r.status === "FAIL");
2317
+ if (fails.length > 0) {
2318
+ out.push("## Open tickets", "", "| Story | Requirement | Evidence |", "| --- | --- | --- |");
2319
+ for (const r of fails) out.push(`| ${escapeCell(r.storyTitle)} | ${escapeCell(r.reqTitle)} | ${ev(r.evidence)} |`);
2320
+ out.push("");
2321
+ } else out.push("_All requirements passing — no open tickets._", "");
2322
+ out.push("## Per-story tick-off", "");
2323
+ for (const storyId of [...new Set(rows.map((r) => r.storyId))]) {
2324
+ const storyRows = rows.filter((r) => r.storyId === storyId);
2325
+ const passed = storyRows.filter((r) => r.status === "PASS").length;
2326
+ const mark = passed === storyRows.length ? "✅" : "⚠️";
2327
+ out.push(`### ${escapeCell(storyRows[0].storyTitle)} — ${passed}/${storyRows.length} ${mark}`, "", "| Requirement | Status | Evidence |", "| --- | --- | --- |");
2328
+ for (const r of storyRows) out.push(`| ${escapeCell(r.reqTitle)} | ${r.status === "PASS" ? "✅ PASS" : "❌ FAIL"} | ${ev(r.evidence)} |`);
2329
+ out.push("");
2330
+ }
2331
+ return out.join("\n");
2332
+ }
2333
+ //#endregion
2334
+ //#region src/campaign/presets/run-profile-matrix.ts
2335
+ /**
2336
+ * `runProfileMatrix` — the missing keystone between `runAgentMatrix` and the
2337
+ * backend-integrity guard.
2338
+ *
2339
+ * The gap it closes: `runAgentMatrix` is a topology-opaque scheduler whose
2340
+ * cells return a bare `{ output, verdict, costUsd }` — no `tokenUsage`, not a
2341
+ * `RunRecord`. `assertRealBackend` / `summarizeBackendIntegrity` key on
2342
+ * `RunRecord.tokenUsage`, so they cannot run on a raw matrix result. Every
2343
+ * consumer therefore hand-writes the same bridge: fan a profile × scenario
2344
+ * cartesian, call dispatch, fabricate a `RunRecord` with token usage, thread it
2345
+ * back, run the integrity guard. That hand-rolled bridge is exactly the pile of
2346
+ * bespoke `eval:*` scripts the adoption skills keep trying (and failing) to
2347
+ * forbid.
2348
+ *
2349
+ * `runProfileMatrix` IS that bridge, once:
2350
+ *
2351
+ * - axis 3 (PROFILE) = `profiles: AgentProfile[]`
2352
+ * - axis 1 (PERSONA/SCENARIO) = `scenarios: Scenario[]` (each scenario carries
2353
+ * its persona; `personaOf` groups them for the `byPersona` pivot)
2354
+ * - the scoring axis = `judges`
2355
+ *
2356
+ * It runs `runCampaign` once per profile (reusing its seeds, reps, bootstrap
2357
+ * CIs, resumability, and the `LabeledScenarioStore` capture flywheel), maps
2358
+ * every cell to a validated `RunRecord` carrying the real `tokenUsage` the
2359
+ * dispatch committed via `ctx.cost.runPaidCall`, and runs `assertRealBackend`
2360
+ * BY CONSTRUCTION before returning — so a stub-backend run fails loudly instead
2361
+ * of reporting a clean 0/N leaderboard.
2362
+ *
2363
+ * Dispatch contract: a dispatch that calls an LLM MUST report usage via
2364
+ * `ctx.cost.runPaidCall({ execute, receipt })`.
2365
+ * A dispatch that reports zero tokens is indistinguishable from a stub and the
2366
+ * integrity guard treats it as one.
2367
+ */
2368
+ /** Thrown when the matrix is misconfigured (no profiles, a profile whose model
2369
+ * lacks a snapshot version, etc.). Distinct from `BackendIntegrityError`,
2370
+ * which signals a stub backend at run time. */
2371
+ var ProfileMatrixError = class extends AgentEvalError {
2372
+ constructor(message) {
2373
+ super("profile_matrix", message);
2374
+ }
2375
+ };
2376
+ function sanitize(id) {
2377
+ return id.replace(/[^a-zA-Z0-9_-]/g, "_");
2378
+ }
2379
+ function sha(input) {
2380
+ return createHash("sha256").update(JSON.stringify(input)).digest("hex");
2381
+ }
2382
+ function mean(xs) {
2383
+ return xs.length === 0 ? 0 : xs.reduce((a, b) => a + b, 0) / xs.length;
2384
+ }
2385
+ /**
2386
+ * Resolve the concrete, snapshot-bearing model for a cell whose profile
2387
+ * declared the `HARNESS_NATIVE_MODEL` sentinel (a vendor-locked harness that
2388
+ * resolves its model at runtime). The dispatch must have reported it via
2389
+ * the model in a paid-call receipt — surfaced as `cell.resolvedModel`. Throws when it is
2390
+ * missing or lacks a snapshot, so a provenance-broken row can never be
2391
+ * recorded as the bare sentinel.
2392
+ */
2393
+ function requireResolvedModel(cell, profileId) {
2394
+ const resolved = cell.resolvedModel?.trim();
2395
+ if (!resolved) throw new ProfileMatrixError(`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' — return it in the ctx.cost.runPaidCall receipt so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`);
2396
+ if (!modelHasSnapshot(resolved)) throw new ProfileMatrixError(`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version — pin it (name@YYYY-MM-DD or name-YYYYMMDD) in the paid-call receipt`);
2397
+ return resolved;
2398
+ }
2399
+ function buildRunRecord(args) {
2400
+ const { cell, profile, profileHash, configHash, experimentId, splitTag, commitSha, matrixId } = args;
2401
+ const profileId = agentProfileId(profile);
2402
+ const declaredModel = agentProfileModelId(profile);
2403
+ const model = declaredModel === "default" ? requireResolvedModel(cell, profileId) : declaredModel;
2404
+ const record = campaignCellToRunRecord(cell, {
2405
+ runId: `${matrixId}:${profileId}:${cell.cellId}`,
2406
+ experimentId,
2407
+ candidateId: profileId,
2408
+ model,
2409
+ promptHash: profileHash,
2410
+ configHash,
2411
+ commitSha,
2412
+ splitTag,
2413
+ agentProfile: args.agentProfileCell
2414
+ });
2415
+ if (args.corpusText && args.scenario) try {
2416
+ const text = args.corpusText(cell.artifact, args.scenario);
2417
+ if (text && typeof text.prompt === "string" && typeof text.completion === "string") {
2418
+ record.prompt = text.prompt;
2419
+ record.completion = text.completion;
2420
+ }
2421
+ } catch {}
2422
+ return record;
2423
+ }
2424
+ /**
2425
+ * Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
2426
+ */
2427
+ async function runProfileMatrix(opts) {
2428
+ if (opts.profiles.length === 0) throw new ProfileMatrixError("profiles must not be empty");
2429
+ if (opts.scenarios.length === 0) throw new ProfileMatrixError("scenarios must not be empty");
2430
+ const splitTag = opts.splitTag ?? "search";
2431
+ const seed = opts.seed ?? 42;
2432
+ const validate = opts.validate ?? true;
2433
+ const integrityMode = opts.integrity ?? "assert";
2434
+ const profileIds = opts.profiles.map(agentProfileId);
2435
+ const experimentId = opts.experimentId ?? `pm_${sha({
2436
+ profileIds,
2437
+ scenarios: opts.scenarios.map((s) => s.id)
2438
+ }).slice(0, 16)}`;
2439
+ const matrixId = `mtx_${sha({
2440
+ experimentId,
2441
+ profileIds,
2442
+ seed,
2443
+ splitTag
2444
+ }).slice(0, 16)}`;
2445
+ const scenarioById = new Map(opts.scenarios.map((s) => [s.id, s]));
2446
+ for (const profile of opts.profiles) {
2447
+ const profileHash = agentProfileHash(profile);
2448
+ const profileId = agentProfileId(profile);
2449
+ const declaredModel = agentProfileModelId(profile);
2450
+ const model = declaredModel === "default" ? `${HARNESS_NATIVE_MODEL}@runtime-resolved` : declaredModel;
2451
+ try {
2452
+ validateRunRecord({
2453
+ runId: `${matrixId}:${profileId}:probe`,
2454
+ experimentId,
2455
+ candidateId: profileId,
2456
+ seed,
2457
+ model,
2458
+ promptHash: profileHash,
2459
+ configHash: profileHash,
2460
+ commitSha: opts.commitSha,
2461
+ wallMs: 0,
2462
+ costUsd: null,
2463
+ costProvenance: {
2464
+ kind: "uncaptured",
2465
+ usd: null
2466
+ },
2467
+ tokenUsage: {
2468
+ input: 0,
2469
+ output: 0
2470
+ },
2471
+ terminalOutcome: "succeeded",
2472
+ outcome: { raw: { execution_error_count: 0 } },
2473
+ splitTag,
2474
+ scenarioId: "recordability-probe"
2475
+ });
2476
+ } catch (err) {
2477
+ throw new ProfileMatrixError(`profile '${profileId}' is not recordable: ${err instanceof Error ? err.message : String(err)}`);
2478
+ }
2479
+ }
2480
+ const records = [];
2481
+ const campaigns = {};
2482
+ const byProfile = {};
2483
+ for (const profile of opts.profiles) {
2484
+ const profileHash = agentProfileHash(profile);
2485
+ const profileId = agentProfileId(profile);
2486
+ const declaredModel = agentProfileModelId(profile);
2487
+ const configHash = sha({
2488
+ profile: profileHash,
2489
+ judges: (opts.judges ?? []).map((j) => j.name),
2490
+ seed,
2491
+ splitTag
2492
+ });
2493
+ const dispatch = (scenario, ctx) => opts.dispatch(profile, scenario, ctx);
2494
+ Object.defineProperty(dispatch, "name", { value: `profile_${sanitize(profileId)}` });
2495
+ const campaign = await runCampaign({
2496
+ scenarios: opts.scenarios,
2497
+ dispatch,
2498
+ judges: opts.judges,
2499
+ seed,
2500
+ reps: opts.reps,
2501
+ maxConcurrency: opts.maxConcurrency,
2502
+ costCeiling: opts.costCeiling,
2503
+ labeledStore: opts.labeledStore,
2504
+ captureSource: opts.captureSource,
2505
+ storage: opts.storage,
2506
+ now: opts.now,
2507
+ runDir: join(opts.runDir, sanitize(profileId))
2508
+ });
2509
+ const axis = harnessAxisOf(profile);
2510
+ const buildCellIdentity = (cellModel) => buildAgentProfileCell({
2511
+ profileId,
2512
+ sourceProfile: {
2513
+ kind: "agent-interface-profile",
2514
+ hash: profileHash
2515
+ },
2516
+ model: cellModel,
2517
+ ...axis ? { harness: { id: axis.harness } } : {}
2518
+ });
2519
+ const sharedCellIdentity = declaredModel === "default" ? void 0 : await buildCellIdentity(declaredModel);
2520
+ const profileRecords = [];
2521
+ for (const cell of campaign.cells) {
2522
+ const agentProfileCell = sharedCellIdentity ?? await buildCellIdentity(requireResolvedModel(cell, profileId));
2523
+ const record = buildRunRecord({
2524
+ cell,
2525
+ profile,
2526
+ profileHash,
2527
+ configHash,
2528
+ experimentId,
2529
+ splitTag,
2530
+ commitSha: opts.commitSha,
2531
+ matrixId,
2532
+ agentProfileCell,
2533
+ scenario: scenarioById.get(cell.scenarioId),
2534
+ corpusText: opts.corpusText
2535
+ });
2536
+ if (validate) validateRunRecord(record);
2537
+ profileRecords.push(record);
2538
+ records.push(record);
2539
+ }
2540
+ const totalCostUsd = campaign.aggregates.cost.totalCostUsd;
2541
+ campaigns[profileId] = campaign;
2542
+ byProfile[profileId] = {
2543
+ profileId,
2544
+ profileHash,
2545
+ model: declaredModel === "default" ? profileRecords[0]?.model ?? declaredModel : declaredModel,
2546
+ records: profileRecords.length,
2547
+ meanComposite: meanOrNull(profileRecords.map(scoreOf).filter((score) => score !== void 0)),
2548
+ totalCostUsd,
2549
+ integrity: summarizeBackendIntegrity(profileRecords)
2550
+ };
2551
+ }
2552
+ const integrity = summarizeBackendIntegrity(records);
2553
+ if (integrityMode === "assert") assertRealBackend(records, { allowMixed: opts.allowMixed ?? true });
2554
+ else if (integrityMode === "warn" && integrity.verdict !== "real") console.warn(`[runProfileMatrix] backend integrity: ${integrity.verdict} — ${integrity.diagnosis}`);
2555
+ return {
2556
+ matrixId,
2557
+ experimentId,
2558
+ records,
2559
+ byProfile,
2560
+ byScenario: rollup(records, (r) => r.scenarioId),
2561
+ byPersona: opts.personaOf ? rollupByPersona(records, opts.scenarios, opts.personaOf) : void 0,
2562
+ integrity,
2563
+ campaigns
2564
+ };
2565
+ }
2566
+ /** Score for a produced RunRecord, absent when the campaign cell was unscored.
2567
+ * Ungated (`runTaskScore` is raw) — a matrix pivot reports measured scores;
2568
+ * it is not a training input. */
2569
+ function scoreOf(r) {
2570
+ return runTaskScore(r);
2571
+ }
2572
+ function meanOrNull(values) {
2573
+ return values.length === 0 ? null : mean(values);
2574
+ }
2575
+ function rollup(records, keyOf) {
2576
+ const groups = /* @__PURE__ */ new Map();
2577
+ for (const r of records) {
2578
+ const key = keyOf(r);
2579
+ if (key === void 0) continue;
2580
+ const score = scoreOf(r);
2581
+ if (score === void 0) continue;
2582
+ const arr = groups.get(key) ?? [];
2583
+ arr.push(score);
2584
+ groups.set(key, arr);
2585
+ }
2586
+ const out = {};
2587
+ for (const [key, xs] of groups) out[key] = {
2588
+ meanComposite: mean(xs),
2589
+ n: xs.length
2590
+ };
2591
+ return out;
2592
+ }
2593
+ function rollupByPersona(records, scenarios, personaOf) {
2594
+ const personaByScenarioId = /* @__PURE__ */ new Map();
2595
+ for (const s of scenarios) personaByScenarioId.set(s.id, personaOf(s));
2596
+ return rollup(records, (r) => r.scenarioId ? personaByScenarioId.get(r.scenarioId) : void 0);
2597
+ }
2598
+ //#endregion
2599
+ //#region src/campaign/scenario-selection.ts
2600
+ const DEFAULT_SATURATION_CEILING = .999;
2601
+ /** Variance below this is treated as "every candidate scored the same". */
2602
+ const VARIANCE_EPSILON = 1e-9;
2603
+ /** Population mean + variance of the candidate scores. Empty ⇒ zeros (a signal
2604
+ * with no observations cannot discriminate). */
2605
+ function moments(scores) {
2606
+ const n = scores.length;
2607
+ if (n === 0) return {
2608
+ meanScore: 0,
2609
+ variance: 0
2610
+ };
2611
+ let sum = 0;
2612
+ for (const s of scores) sum += s;
2613
+ const meanScore = sum / n;
2614
+ let sqDev = 0;
2615
+ for (const s of scores) {
2616
+ const d = s - meanScore;
2617
+ sqDev += d * d;
2618
+ }
2619
+ return {
2620
+ meanScore,
2621
+ variance: sqDev / n
2622
+ };
2623
+ }
2624
+ /** Deterministic ordering: discrimination desc, then meanScore asc (more
2625
+ * headroom first), then scenarioId asc. */
2626
+ function compareDiscrimination(a, b) {
2627
+ if (b.discrimination !== a.discrimination) return b.discrimination - a.discrimination;
2628
+ if (a.meanScore !== b.meanScore) return a.meanScore - b.meanScore;
2629
+ return a.scenarioId < b.scenarioId ? -1 : a.scenarioId > b.scenarioId ? 1 : 0;
2630
+ }
2631
+ /**
2632
+ * Rank scenarios by how well they DISCRIMINATE candidates.
2633
+ *
2634
+ * `discrimination = variance` (spread of the candidate scores) — kept simple on
2635
+ * purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
2636
+ * so that, among equally spread scenarios, the one with more room to improve
2637
+ * ranks first. Returned sorted by the deterministic order above.
2638
+ */
2639
+ function scoreDiscrimination(signals, opts) {
2640
+ const saturationCeiling = opts?.saturationCeiling ?? DEFAULT_SATURATION_CEILING;
2641
+ return signals.map((signal) => {
2642
+ const { meanScore, variance } = moments(signal.scores);
2643
+ const tied = variance < VARIANCE_EPSILON && meanScore >= saturationCeiling;
2644
+ return {
2645
+ scenarioId: signal.scenarioId,
2646
+ discrimination: variance,
2647
+ meanScore,
2648
+ variance,
2649
+ tied
2650
+ };
2651
+ }).sort(compareDiscrimination);
2652
+ }
2653
+ /**
2654
+ * Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
2655
+ * fully saturated ties when enough non-tied scenarios exist (a tie in the
2656
+ * holdout wastes a paired cell).
2657
+ *
2658
+ * Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
2659
+ * least-saturated tied ones (tied scenarios are already ordered least-saturated
2660
+ * first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
2661
+ * `signals.length <= k`, returns all ids in discrimination order.
2662
+ */
2663
+ function selectDiscriminative(signals, k, opts) {
2664
+ if (k < 1) throw new Error(`selectDiscriminative: k must be >= 1 (got ${k})`);
2665
+ const ranked = scoreDiscrimination(signals, opts);
2666
+ if (ranked.length <= k) return ranked.map((s) => s.scenarioId);
2667
+ const nonTied = ranked.filter((s) => !s.tied);
2668
+ if (nonTied.length >= k) return nonTied.slice(0, k).map((s) => s.scenarioId);
2669
+ const fill = ranked.filter((s) => s.tied).slice(0, k - nonTied.length);
2670
+ return [...nonTied, ...fill].map((s) => s.scenarioId);
2671
+ }
2672
+ //#endregion
2673
+ //#region src/campaign/search-ledger.ts
2674
+ /**
2675
+ * Durable append-only audit log for improvement searches.
2676
+ *
2677
+ * Existing campaign artifacts keep their own rich records: `RunRecord` owns a
2678
+ * measured run and `CostLedger` owns per-call accounting. This ledger does not
2679
+ * copy those structures. It binds their immutable ids and receipts into one replayable event stream so a
2680
+ * search can answer, after a crash, exactly which candidates and task attempts
2681
+ * existed, which surfaces actually fired, what they cost, and why they were
2682
+ * selected or rejected.
2683
+ *
2684
+ * The file format is canonical JSONL with a SHA-256 hash chain. Every append is
2685
+ * serialized across processes, fsynced before acknowledgement, and idempotent
2686
+ * by `eventId`. A malformed, non-canonical, truncated, reordered, or conflicting
2687
+ * log fails loudly; the implementation never skips a bad row.
2688
+ *
2689
+ * The journal machinery itself (hash chain, locking, fsync, idempotent append)
2690
+ * is the generic `ledger-core` journal; this module supplies the campaign
2691
+ * codec: event schemas, canonical event ordering, and the search state machine.
2692
+ */
2693
+ const SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
2694
+ const NON_EMPTY = z.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
2695
+ const HASH = z.string().regex(/^sha256:[a-f0-9]{64}$/);
2696
+ const LINEAGE_NODE_ID = z.string().regex(/^[a-f0-9]{16}$/);
2697
+ const IMMUTABLE_REVISION = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
2698
+ const ISO_TIMESTAMP = z.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
2699
+ const NON_NEGATIVE_INT = z.number().int().nonnegative().safe();
2700
+ const FINITE_NUMBER = z.number().finite();
2701
+ const ArtifactRefSchema = z.object({
2702
+ role: NON_EMPTY,
2703
+ uri: NON_EMPTY,
2704
+ sha256: HASH,
2705
+ byteLength: NON_NEGATIVE_INT
2706
+ }).strict();
2707
+ const SourceRefSchema = z.object({
2708
+ uri: NON_EMPTY,
2709
+ revision: IMMUTABLE_REVISION
2710
+ }).strict();
2711
+ const FailureReasonSchema = z.object({
2712
+ code: NON_EMPTY,
2713
+ message: NON_EMPTY
2714
+ }).strict();
2715
+ const EventBaseShape = {
2716
+ eventId: NON_EMPTY,
2717
+ occurredAt: ISO_TIMESTAMP,
2718
+ artifacts: z.array(ArtifactRefSchema).min(1)
2719
+ };
2720
+ const OperationKindSchema = z.enum([
2721
+ "candidate-generation",
2722
+ "analysis",
2723
+ "selection",
2724
+ "judge",
2725
+ "other"
2726
+ ]);
2727
+ const SearchPlannedSchema = z.object({
2728
+ ...EventBaseShape,
2729
+ kind: z.literal("search-planned"),
2730
+ plan: z.object({
2731
+ candidateSlots: z.array(z.object({
2732
+ slotId: NON_EMPTY,
2733
+ generationOperationId: NON_EMPTY
2734
+ }).strict()).min(1),
2735
+ tasks: z.array(z.object({
2736
+ taskId: NON_EMPTY,
2737
+ source: SourceRefSchema,
2738
+ benchmark: SourceRefSchema,
2739
+ maxAttempts: z.number().int().positive().safe()
2740
+ }).strict()).min(1),
2741
+ operations: z.array(z.object({
2742
+ operationId: NON_EMPTY,
2743
+ kind: OperationKindSchema
2744
+ }).strict()).min(1)
2745
+ }).strict()
2746
+ }).strict();
2747
+ const CandidateRegisteredSchema = z.object({
2748
+ ...EventBaseShape,
2749
+ kind: z.literal("candidate-registered"),
2750
+ slotId: NON_EMPTY,
2751
+ generationOperationId: NON_EMPTY,
2752
+ candidateId: NON_EMPTY,
2753
+ lineage: z.object({
2754
+ lineageNodeId: LINEAGE_NODE_ID,
2755
+ parentCandidateIds: z.array(NON_EMPTY),
2756
+ generation: NON_NEGATIVE_INT,
2757
+ proposer: NON_EMPTY,
2758
+ proposerSource: SourceRefSchema
2759
+ }).strict(),
2760
+ surfaces: z.array(z.object({
2761
+ surfaceId: NON_EMPTY,
2762
+ kind: z.enum([
2763
+ "prompt",
2764
+ "tool-contract",
2765
+ "runtime-config",
2766
+ "memory",
2767
+ "knowledge",
2768
+ "agent-profile",
2769
+ "code",
2770
+ "deployment"
2771
+ ]),
2772
+ artifact: ArtifactRefSchema
2773
+ }).strict()).min(1)
2774
+ }).strict();
2775
+ const CandidateSlotClosedSchema = z.object({
2776
+ ...EventBaseShape,
2777
+ kind: z.literal("candidate-slot-closed"),
2778
+ slotId: NON_EMPTY,
2779
+ generationOperationId: NON_EMPTY,
2780
+ reason: FailureReasonSchema
2781
+ }).strict();
2782
+ const KnownTokensSchema = z.object({
2783
+ status: z.literal("known"),
2784
+ inputTokens: NON_NEGATIVE_INT,
2785
+ outputTokens: NON_NEGATIVE_INT,
2786
+ cachedTokens: NON_NEGATIVE_INT
2787
+ }).strict();
2788
+ const UnknownSchema = z.object({
2789
+ status: z.literal("unknown"),
2790
+ reason: NON_EMPTY
2791
+ }).strict();
2792
+ const KnownCostSchema = z.object({
2793
+ status: z.literal("known"),
2794
+ usd: z.number().finite().nonnegative(),
2795
+ source: z.enum([
2796
+ "provider",
2797
+ "pricing-table",
2798
+ "free"
2799
+ ])
2800
+ }).strict().superRefine((cost, ctx) => {
2801
+ if (cost.source === "free" && cost.usd !== 0) ctx.addIssue({
2802
+ code: "custom",
2803
+ message: "free cost source must have usd 0"
2804
+ });
2805
+ });
2806
+ const UnknownCostSchema = z.object({
2807
+ status: z.literal("unknown"),
2808
+ knownLowerBoundUsd: z.number().finite().nonnegative(),
2809
+ reason: NON_EMPTY
2810
+ }).strict();
2811
+ const AccountingSchema = z.object({
2812
+ tokens: z.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
2813
+ cost: z.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
2814
+ }).strict();
2815
+ const MetricsSchema = z.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
2816
+ for (const key of Object.keys(metrics)) if (key === "__proto__" || key === "constructor" || key === "prototype") ctx.addIssue({
2817
+ code: "custom",
2818
+ message: `unsafe metric key ${key}`
2819
+ });
2820
+ });
2821
+ const OutcomeSchema = z.discriminatedUnion("status", [
2822
+ z.object({
2823
+ status: z.literal("passed"),
2824
+ score: FINITE_NUMBER,
2825
+ metrics: MetricsSchema
2826
+ }).strict(),
2827
+ z.object({
2828
+ status: z.literal("failed"),
2829
+ score: FINITE_NUMBER,
2830
+ metrics: MetricsSchema,
2831
+ failure: FailureReasonSchema
2832
+ }).strict(),
2833
+ z.object({
2834
+ status: z.literal("errored"),
2835
+ metrics: MetricsSchema,
2836
+ error: FailureReasonSchema.extend({ retryable: z.boolean() }).strict()
2837
+ }).strict()
2838
+ ]);
2839
+ const EffectSchema = z.discriminatedUnion("status", [z.object({
2840
+ status: z.literal("measured"),
2841
+ metric: NON_EMPTY,
2842
+ baselineValue: FINITE_NUMBER,
2843
+ candidateValue: FINITE_NUMBER,
2844
+ delta: FINITE_NUMBER
2845
+ }).strict().superRefine((effect, ctx) => {
2846
+ const expected = effect.candidateValue - effect.baselineValue;
2847
+ const tolerance = Number.EPSILON * Math.max(1, Math.abs(expected), Math.abs(effect.delta)) * 8;
2848
+ if (Math.abs(effect.delta - expected) > tolerance) ctx.addIssue({
2849
+ code: "custom",
2850
+ message: "delta must equal candidateValue - baselineValue"
2851
+ });
2852
+ }), z.object({
2853
+ status: z.literal("not-measured"),
2854
+ reason: NON_EMPTY
2855
+ }).strict()]);
2856
+ const SurfaceEvidenceSchema = z.object({
2857
+ surfaceId: NON_EMPTY,
2858
+ fired: z.boolean(),
2859
+ firingCount: NON_NEGATIVE_INT,
2860
+ effect: EffectSchema,
2861
+ evidence: z.array(ArtifactRefSchema).min(1)
2862
+ }).strict().superRefine((evidence, ctx) => {
2863
+ if (evidence.fired && evidence.firingCount === 0) ctx.addIssue({
2864
+ code: "custom",
2865
+ message: "a fired surface must have firingCount >= 1"
2866
+ });
2867
+ if (!evidence.fired && evidence.firingCount !== 0) ctx.addIssue({
2868
+ code: "custom",
2869
+ message: "a surface that did not fire must have firingCount 0"
2870
+ });
2871
+ if (!evidence.fired && evidence.effect.status === "measured" && evidence.effect.delta !== 0) ctx.addIssue({
2872
+ code: "custom",
2873
+ message: "a surface that did not fire cannot claim non-zero effect"
2874
+ });
2875
+ });
2876
+ const TaskAttemptedSchema = z.object({
2877
+ ...EventBaseShape,
2878
+ kind: z.literal("task-attempted"),
2879
+ candidateId: NON_EMPTY,
2880
+ runId: NON_EMPTY,
2881
+ attemptIndex: NON_NEGATIVE_INT,
2882
+ task: z.object({
2883
+ taskId: NON_EMPTY,
2884
+ source: SourceRefSchema
2885
+ }).strict(),
2886
+ identity: z.object({
2887
+ model: z.object({
2888
+ provider: NON_EMPTY,
2889
+ snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
2890
+ }).strict(),
2891
+ agent: SourceRefSchema,
2892
+ benchmark: SourceRefSchema
2893
+ }).strict(),
2894
+ outcome: OutcomeSchema,
2895
+ accounting: AccountingSchema,
2896
+ surfaceEvidence: z.array(SurfaceEvidenceSchema).min(1)
2897
+ }).strict();
2898
+ const SearchOperationRecordedSchema = z.object({
2899
+ ...EventBaseShape,
2900
+ kind: z.literal("search-operation-recorded"),
2901
+ operationId: NON_EMPTY,
2902
+ operationKind: OperationKindSchema,
2903
+ execution: z.discriminatedUnion("kind", [z.object({
2904
+ kind: z.literal("model"),
2905
+ model: z.object({
2906
+ provider: NON_EMPTY,
2907
+ snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
2908
+ }).strict(),
2909
+ source: SourceRefSchema
2910
+ }).strict(), z.object({
2911
+ kind: z.literal("deterministic"),
2912
+ source: SourceRefSchema
2913
+ }).strict()]),
2914
+ outcome: z.discriminatedUnion("status", [
2915
+ z.object({ status: z.literal("completed") }).strict(),
2916
+ z.object({
2917
+ status: z.literal("partial"),
2918
+ failure: FailureReasonSchema
2919
+ }).strict(),
2920
+ z.object({
2921
+ status: z.literal("failed"),
2922
+ failure: FailureReasonSchema
2923
+ }).strict()
2924
+ ]),
2925
+ accounting: AccountingSchema
2926
+ }).strict();
2927
+ const CandidateDecidedSchema = z.object({
2928
+ ...EventBaseShape,
2929
+ kind: z.literal("candidate-decided"),
2930
+ candidateId: NON_EMPTY,
2931
+ decision: z.discriminatedUnion("status", [z.object({ status: z.literal("selected") }).strict(), z.object({
2932
+ status: z.literal("rejected"),
2933
+ reason: FailureReasonSchema
2934
+ }).strict()])
2935
+ }).strict();
2936
+ const SearchCompletedSchema = z.object({
2937
+ ...EventBaseShape,
2938
+ kind: z.literal("search-completed"),
2939
+ result: z.discriminatedUnion("status", [z.object({
2940
+ status: z.literal("selected"),
2941
+ candidateId: NON_EMPTY
2942
+ }).strict(), z.object({
2943
+ status: z.literal("all-rejected"),
2944
+ reason: FailureReasonSchema
2945
+ }).strict()])
2946
+ }).strict();
2947
+ const EventSchema = z.discriminatedUnion("kind", [
2948
+ SearchPlannedSchema,
2949
+ CandidateRegisteredSchema,
2950
+ CandidateSlotClosedSchema,
2951
+ TaskAttemptedSchema,
2952
+ SearchOperationRecordedSchema,
2953
+ CandidateDecidedSchema,
2954
+ SearchCompletedSchema
2955
+ ]);
2956
+ const EntrySchema = z.object({
2957
+ schema: z.literal(SEARCH_LEDGER_SCHEMA),
2958
+ campaignId: NON_EMPTY,
2959
+ sequence: NON_NEGATIVE_INT,
2960
+ previousHash: z.union([HASH, z.null()]),
2961
+ event: EventSchema,
2962
+ entryHash: HASH
2963
+ }).strict();
2964
+ /** Validate and return a canonical copy. Arrays whose order is not semantic are
2965
+ * sorted so retries from different processes produce byte-identical events. */
2966
+ function validateSearchLedgerEvent(input) {
2967
+ const parsed = EventSchema.safeParse(input);
2968
+ if (!parsed.success) throw new SearchLedgerError(`invalid search ledger event: ${formatZodError(parsed.error)}`);
2969
+ return normalizeEvent(parsed.data);
2970
+ }
2971
+ /** Open a durable filesystem search ledger. Construction performs no I/O; the
2972
+ * first `append` or `replay` validates the complete existing file. */
2973
+ function openSearchLedger(options) {
2974
+ if (options.path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
2975
+ return new FileSearchLedger(options.path, options.campaignId);
2976
+ }
2977
+ function searchLedgerCodec(campaignId) {
2978
+ return {
2979
+ ...SEARCH_LEDGER_FILE_CONTEXT,
2980
+ header: {
2981
+ schema: SEARCH_LEDGER_SCHEMA,
2982
+ campaignId
2983
+ },
2984
+ conflictError: (message) => new SearchLedgerConflictError(message),
2985
+ parseEntry: parseSearchLedgerEntry,
2986
+ checkEntryHeader: (entry, index) => {
2987
+ if (entry.campaignId !== campaignId) throw new SearchLedgerIntegrityError(`entry ${index} belongs to campaign ${entry.campaignId}, expected ${campaignId}`);
2988
+ },
2989
+ createProjector: () => createSearchLedgerProjector(campaignId)
2990
+ };
2991
+ }
2992
+ var FileSearchLedger = class {
2993
+ path;
2994
+ campaignId;
2995
+ journal;
2996
+ constructor(path, campaignId) {
2997
+ if (path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
2998
+ if (campaignId.length === 0) throw new SearchLedgerError("campaignId is empty");
2999
+ if (campaignId.trim() !== campaignId) throw new SearchLedgerError("campaignId must not contain surrounding whitespace");
3000
+ this.campaignId = campaignId;
3001
+ this.journal = new FileLedgerJournal(path, searchLedgerCodec(campaignId));
3002
+ this.path = this.journal.path;
3003
+ }
3004
+ async replay() {
3005
+ return (await this.journal.replay()).projection;
3006
+ }
3007
+ async append(input) {
3008
+ const event = validateSearchLedgerEvent(input);
3009
+ const { entry, appended, projection } = await this.journal.append(event);
3010
+ return {
3011
+ entry,
3012
+ appended,
3013
+ replay: projection
3014
+ };
3015
+ }
3016
+ };
3017
+ function parseSearchLedgerEntry(raw, context) {
3018
+ const parsed = EntrySchema.safeParse(raw);
3019
+ if (!parsed.success) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has a malformed entry at line ${context.line}: ${formatZodError(parsed.error)}`);
3020
+ const entry = parsed.data;
3021
+ if (canonicalString(validateSearchLedgerEvent(entry.event)) !== canonicalString(entry.event)) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has non-canonical event ordering at line ${context.line}`);
3022
+ return entry;
3023
+ }
3024
+ /** Replay the campaign search state machine over chain-verified entries. The
3025
+ * generic journal owns sequence, hash, and eventId-uniqueness checks; this
3026
+ * projector owns every campaign invariant and builds the replay projection. */
3027
+ function createSearchLedgerProjector(campaignId) {
3028
+ const candidates = /* @__PURE__ */ new Map();
3029
+ const candidateBySlot = /* @__PURE__ */ new Map();
3030
+ const closedSlots = /* @__PURE__ */ new Map();
3031
+ const lineageNodes = /* @__PURE__ */ new Map();
3032
+ const runIds = /* @__PURE__ */ new Set();
3033
+ const attemptKeys = /* @__PURE__ */ new Set();
3034
+ const candidateEvents = [];
3035
+ const closedSlotEvents = [];
3036
+ const attempts = [];
3037
+ const operationEvents = [];
3038
+ const operationsById = /* @__PURE__ */ new Map();
3039
+ const decisions = [];
3040
+ let planEvent = null;
3041
+ let completion = null;
3042
+ let previousOccurredAt = Number.NEGATIVE_INFINITY;
3043
+ const apply = (entry, index) => {
3044
+ const event = entry.event;
3045
+ if (completion) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears after terminal event ${completion.eventId}`);
3046
+ const occurredAt = Date.parse(event.occurredAt);
3047
+ if (occurredAt < previousOccurredAt) throw new SearchLedgerIntegrityError(`event ${event.eventId} occurred before the preceding durable event`);
3048
+ previousOccurredAt = occurredAt;
3049
+ assertUnique(event.artifacts.map(artifactKey), "artifact receipt", event.eventId);
3050
+ if (event.kind === "search-planned") {
3051
+ if (index !== 0 || planEvent) throw new SearchLedgerIntegrityError("search plan must be the first and only plan event");
3052
+ assertUnique(event.plan.candidateSlots.map((slot) => slot.slotId), "candidate slot", event.eventId);
3053
+ assertUnique(event.plan.tasks.map((task) => task.taskId), "planned taskId", event.eventId);
3054
+ assertUnique(event.plan.operations.map((operation) => operation.operationId), "planned operationId", event.eventId);
3055
+ for (const slot of event.plan.candidateSlots) if (event.plan.operations.find((operation) => operation.operationId === slot.generationOperationId)?.kind !== "candidate-generation") throw new SearchLedgerIntegrityError(`candidate slot ${slot.slotId} references unplanned candidate-generation operation ${slot.generationOperationId}`);
3056
+ planEvent = event;
3057
+ return;
3058
+ }
3059
+ if (!planEvent) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears before the required search plan`);
3060
+ if (event.kind === "candidate-registered") {
3061
+ if (candidates.has(event.candidateId)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was registered twice`);
3062
+ const plannedSlot = planEvent.plan.candidateSlots.find((slot) => slot.slotId === event.slotId);
3063
+ if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} binds unknown slot ${event.slotId}`);
3064
+ if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was bound twice`);
3065
+ if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already closed`);
3066
+ if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
3067
+ const generationOperation = operationsById.get(event.generationOperationId);
3068
+ if (!generationOperation) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} precedes generation operation ${event.generationOperationId}`);
3069
+ if (generationOperation.outcome.status === "failed") throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot bind failed generation operation ${event.generationOperationId}`);
3070
+ const previousCandidate = lineageNodes.get(event.lineage.lineageNodeId);
3071
+ if (previousCandidate) throw new SearchLedgerIntegrityError(`lineage node ${event.lineage.lineageNodeId} is already bound to ${previousCandidate}`);
3072
+ assertUnique(event.lineage.parentCandidateIds, "parentCandidateId", event.eventId);
3073
+ const parents = event.lineage.parentCandidateIds.map((id) => {
3074
+ const parent = candidates.get(id);
3075
+ if (!parent) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} references unknown parent ${id}`);
3076
+ return parent;
3077
+ });
3078
+ const expectedGeneration = parents.length === 0 ? 0 : Math.max(...parents.map((parent) => parent.registered.lineage.generation)) + 1;
3079
+ if (event.lineage.generation !== expectedGeneration) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation ${event.lineage.generation} does not follow its parents (expected ${expectedGeneration})`);
3080
+ assertUnique(event.surfaces.map((surface) => surface.surfaceId), "surfaceId", event.eventId);
3081
+ candidates.set(event.candidateId, {
3082
+ registered: event,
3083
+ attempts: [],
3084
+ decision: null
3085
+ });
3086
+ candidateBySlot.set(event.slotId, event.candidateId);
3087
+ lineageNodes.set(event.lineage.lineageNodeId, event.candidateId);
3088
+ candidateEvents.push(event);
3089
+ return;
3090
+ }
3091
+ if (event.kind === "task-attempted") {
3092
+ const candidate = candidates.get(event.candidateId);
3093
+ if (!candidate) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unknown candidate ${event.candidateId}`);
3094
+ if (candidate.decision) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} appears after candidate ${event.candidateId} was decided`);
3095
+ const plannedTask = planEvent.plan.tasks.find((task) => task.taskId === event.task.taskId);
3096
+ if (!plannedTask) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unplanned task ${event.task.taskId}`);
3097
+ if (canonicalString(plannedTask.source) !== canonicalString(event.task.source) || canonicalString(plannedTask.benchmark) !== canonicalString(event.identity.benchmark)) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} does not match its planned source identity`);
3098
+ if (event.attemptIndex >= plannedTask.maxAttempts) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} attempt ${event.attemptIndex} exceeds planned maxAttempts ${plannedTask.maxAttempts}`);
3099
+ if (runIds.has(event.runId)) throw new SearchLedgerIntegrityError(`runId ${event.runId} was recorded twice`);
3100
+ runIds.add(event.runId);
3101
+ const attemptKey = canonicalString([
3102
+ event.candidateId,
3103
+ event.task.taskId,
3104
+ event.attemptIndex
3105
+ ]);
3106
+ if (attemptKeys.has(attemptKey)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt ${event.attemptIndex} was recorded twice`);
3107
+ const expectedAttemptIndex = candidate.attempts.filter((attempt) => attempt.task.taskId === event.task.taskId).length;
3108
+ if (event.attemptIndex !== expectedAttemptIndex) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt index ${event.attemptIndex} is not contiguous (expected ${expectedAttemptIndex})`);
3109
+ const previousAttempt = candidate.attempts.find((attempt) => attempt.task.taskId === event.task.taskId);
3110
+ if (previousAttempt?.outcome.status !== void 0 && previousAttempt.outcome.status !== "errored") throw new SearchLedgerIntegrityError(`task ${event.task.taskId} was retried after a measured outcome`);
3111
+ if (previousAttempt && canonicalString({
3112
+ task: previousAttempt.task,
3113
+ identity: previousAttempt.identity
3114
+ }) !== canonicalString({
3115
+ task: event.task,
3116
+ identity: event.identity
3117
+ })) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} changed immutable execution identity between attempts`);
3118
+ attemptKeys.add(attemptKey);
3119
+ const declared = candidate.registered.surfaces.map((surface) => surface.surfaceId).sort();
3120
+ const observed = event.surfaceEvidence.map((surface) => surface.surfaceId).sort();
3121
+ assertUnique(observed, "surface evidence", event.eventId);
3122
+ for (const evidence of event.surfaceEvidence) assertUnique(evidence.evidence.map(artifactKey), "surface evidence receipt", event.eventId);
3123
+ if (canonicalString(declared) !== canonicalString(observed)) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} surface evidence does not exactly cover candidate ${event.candidateId}`);
3124
+ candidate.attempts.push(event);
3125
+ attempts.push(event);
3126
+ return;
3127
+ }
3128
+ if (event.kind === "search-operation-recorded") {
3129
+ const plannedOperation = planEvent.plan.operations.find((operation) => operation.operationId === event.operationId);
3130
+ if (!plannedOperation) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was not declared in the search plan`);
3131
+ if (plannedOperation.kind !== event.operationKind) throw new SearchLedgerIntegrityError(`operation ${event.operationId} kind ${event.operationKind} does not match planned ${plannedOperation.kind}`);
3132
+ if (operationsById.has(event.operationId)) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was recorded twice`);
3133
+ operationsById.set(event.operationId, event);
3134
+ operationEvents.push(event);
3135
+ return;
3136
+ }
3137
+ if (event.kind === "candidate-slot-closed") {
3138
+ const plannedSlot = planEvent.plan.candidateSlots.find((slot) => slot.slotId === event.slotId);
3139
+ if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} references unknown slot ${event.slotId}`);
3140
+ if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
3141
+ if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already bound to a candidate`);
3142
+ if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was closed twice`);
3143
+ const operation = operationsById.get(event.generationOperationId);
3144
+ if (!operation) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} precedes operation ${event.generationOperationId}`);
3145
+ if (operation.outcome.status === "completed") throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} cannot close from completed operation ${event.generationOperationId}`);
3146
+ closedSlots.set(event.slotId, event);
3147
+ closedSlotEvents.push(event);
3148
+ return;
3149
+ }
3150
+ if (event.kind === "candidate-decided") {
3151
+ const candidate = candidates.get(event.candidateId);
3152
+ if (!candidate) throw new SearchLedgerIntegrityError(`decision ${event.eventId} references unknown candidate ${event.candidateId}`);
3153
+ if (candidate.decision) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was decided twice`);
3154
+ if (event.decision.status === "selected") {
3155
+ if (!candidate.attempts.some((attempt) => attempt.outcome.status !== "errored")) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot be selected without a measured task outcome`);
3156
+ if (decisions.some((decision) => decision.decision.status === "selected")) throw new SearchLedgerIntegrityError("more than one candidate was selected");
3157
+ }
3158
+ candidate.decision = event;
3159
+ decisions.push(event);
3160
+ return;
3161
+ }
3162
+ const missingCandidateSlots = planEvent.plan.candidateSlots.filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId);
3163
+ if (missingCandidateSlots.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing candidate slots: ${missingCandidateSlots.join(", ")}`);
3164
+ const missingTaskOutcomes = plannedTaskOutcomeKeys(planEvent, candidates);
3165
+ if (missingTaskOutcomes.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing task outcomes: ${missingTaskOutcomes.join(", ")}`);
3166
+ const missingOperations = planEvent.plan.operations.filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId);
3167
+ if (missingOperations.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing search operations: ${missingOperations.join(", ")}`);
3168
+ for (const operation of planEvent.plan.operations) {
3169
+ if (operation.kind !== "candidate-generation") continue;
3170
+ const generationOutcome = operationsById.get(operation.operationId).outcome.status;
3171
+ const slots = planEvent.plan.candidateSlots.filter((slot) => slot.generationOperationId === operation.operationId);
3172
+ if (slots.length === 0) continue;
3173
+ const registeredCount = slots.filter((slot) => candidateBySlot.has(slot.slotId)).length;
3174
+ const closedCount = slots.filter((slot) => closedSlots.has(slot.slotId)).length;
3175
+ if (generationOutcome === "completed" && closedCount > 0) throw new SearchLedgerIntegrityError(`completed generation operation ${operation.operationId} contains ${closedCount} closed slot(s)`);
3176
+ if (generationOutcome === "failed" && registeredCount > 0) throw new SearchLedgerIntegrityError(`failed generation operation ${operation.operationId} contains ${registeredCount} registered candidate(s)`);
3177
+ if (generationOutcome === "partial" && (registeredCount === 0 || closedCount === 0)) throw new SearchLedgerIntegrityError(`partial generation operation ${operation.operationId} must contain both a registered candidate and a closed slot`);
3178
+ }
3179
+ const pending = [...candidates.values()].filter((candidate) => candidate.decision === null);
3180
+ if (pending.length > 0) throw new SearchLedgerIntegrityError(`search completed with ${pending.length} candidate decision(s) missing`);
3181
+ const selected = decisions.filter((decision) => decision.decision.status === "selected");
3182
+ if (event.result.status === "selected") {
3183
+ if (selected.length !== 1 || selected[0].candidateId !== event.result.candidateId) throw new SearchLedgerIntegrityError(`search completion winner ${event.result.candidateId} does not match candidate decisions`);
3184
+ } else if (selected.length !== 0) throw new SearchLedgerIntegrityError("all-rejected completion contains a selected candidate");
3185
+ completion = event;
3186
+ };
3187
+ const finish = (entries) => {
3188
+ const selectedDecisions = decisions.filter((decision) => decision.decision.status === "selected");
3189
+ const rejectedDecisions = decisions.filter((decision) => decision.decision.status === "rejected");
3190
+ const outcomeCounts = {
3191
+ passed: 0,
3192
+ failed: 0,
3193
+ errored: 0
3194
+ };
3195
+ const operationOutcomeCounts = {
3196
+ completed: 0,
3197
+ partial: 0,
3198
+ failed: 0
3199
+ };
3200
+ let inputTokens = 0;
3201
+ let outputTokens = 0;
3202
+ let cachedTokens = 0;
3203
+ let costUsd = 0;
3204
+ const unknownTokenEventIds = [];
3205
+ const unknownCostEventIds = [];
3206
+ for (const attempt of attempts) outcomeCounts[attempt.outcome.status] += 1;
3207
+ for (const operation of operationEvents) operationOutcomeCounts[operation.outcome.status] += 1;
3208
+ for (const costedEvent of [...attempts, ...operationEvents]) {
3209
+ if (costedEvent.accounting.tokens.status === "known") {
3210
+ inputTokens += costedEvent.accounting.tokens.inputTokens;
3211
+ outputTokens += costedEvent.accounting.tokens.outputTokens;
3212
+ cachedTokens += costedEvent.accounting.tokens.cachedTokens;
3213
+ } else unknownTokenEventIds.push(costedEvent.eventId);
3214
+ if (costedEvent.accounting.cost.status === "known") costUsd += costedEvent.accounting.cost.usd;
3215
+ else {
3216
+ costUsd += costedEvent.accounting.cost.knownLowerBoundUsd;
3217
+ unknownCostEventIds.push(costedEvent.eventId);
3218
+ }
3219
+ }
3220
+ const accounting = unknownTokenEventIds.length === 0 && unknownCostEventIds.length === 0 ? {
3221
+ status: "known",
3222
+ inputTokens,
3223
+ outputTokens,
3224
+ cachedTokens,
3225
+ costUsd
3226
+ } : {
3227
+ status: "partial",
3228
+ knownInputTokens: inputTokens,
3229
+ knownOutputTokens: outputTokens,
3230
+ knownCachedTokens: cachedTokens,
3231
+ knownCostUsd: costUsd,
3232
+ unknownTokenEventIds,
3233
+ unknownCostEventIds
3234
+ };
3235
+ const selectedCandidateId = completion?.result.status === "selected" ? completion.result.candidateId : null;
3236
+ const status = completion?.result.status === "selected" ? "selected" : completion?.result.status === "all-rejected" ? "all-rejected" : "in-progress";
3237
+ const missingCandidateSlots = planEvent?.plan.candidateSlots.filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId) ?? [];
3238
+ const missingTaskOutcomes = planEvent ? plannedTaskOutcomeKeys(planEvent, candidates) : [];
3239
+ const missingOperations = planEvent?.plan.operations.filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId) ?? [];
3240
+ return {
3241
+ entries: [...entries],
3242
+ plan: planEvent,
3243
+ candidates: candidateEvents,
3244
+ closedCandidateSlots: closedSlotEvents,
3245
+ attempts,
3246
+ operations: operationEvents,
3247
+ decisions,
3248
+ completion,
3249
+ audit: {
3250
+ campaignId,
3251
+ eventCount: entries.length,
3252
+ candidateCount: candidates.size,
3253
+ closedCandidateSlotCount: closedSlots.size,
3254
+ attemptCount: attempts.length,
3255
+ operationCount: operationEvents.length,
3256
+ outcomes: outcomeCounts,
3257
+ operationOutcomes: operationOutcomeCounts,
3258
+ decisions: {
3259
+ selected: selectedDecisions.length,
3260
+ rejected: rejectedDecisions.length,
3261
+ pending: candidates.size - decisions.length
3262
+ },
3263
+ expected: {
3264
+ candidateSlots: planEvent?.plan.candidateSlots.length ?? 0,
3265
+ taskOutcomes: candidates.size * (planEvent?.plan.tasks.length ?? 0),
3266
+ operations: planEvent?.plan.operations.length ?? 0,
3267
+ missingCandidateSlots,
3268
+ missingTaskOutcomes,
3269
+ missingOperations
3270
+ },
3271
+ status,
3272
+ selectedCandidateId,
3273
+ accounting,
3274
+ headHash: entries.at(-1)?.entryHash ?? null
3275
+ }
3276
+ };
3277
+ };
3278
+ return {
3279
+ apply,
3280
+ finish
3281
+ };
3282
+ }
3283
+ function plannedTaskOutcomeKeys(planEvent, candidates) {
3284
+ const missing = [];
3285
+ const registeredCandidates = [...candidates.values()].sort((a, b) => compareStrings(a.registered.slotId, b.registered.slotId));
3286
+ for (const candidate of registeredCandidates) {
3287
+ const slotId = candidate.registered.slotId;
3288
+ for (const task of planEvent.plan.tasks) if (!candidate.attempts.some((attempt) => attempt.task.taskId === task.taskId && attempt.outcome.status !== "errored")) missing.push(`${slotId}/${task.taskId}`);
3289
+ }
3290
+ return missing;
3291
+ }
3292
+ function normalizeEvent(event) {
3293
+ const artifacts = sortArtifacts(event.artifacts);
3294
+ if (event.kind === "search-planned") return {
3295
+ ...event,
3296
+ artifacts,
3297
+ plan: {
3298
+ candidateSlots: [...event.plan.candidateSlots].sort((a, b) => compareStrings(a.slotId, b.slotId)),
3299
+ tasks: [...event.plan.tasks].sort((a, b) => compareStrings(a.taskId, b.taskId)),
3300
+ operations: [...event.plan.operations].sort((a, b) => compareStrings(a.operationId, b.operationId))
3301
+ }
3302
+ };
3303
+ if (event.kind === "candidate-registered") return {
3304
+ ...event,
3305
+ artifacts,
3306
+ lineage: {
3307
+ ...event.lineage,
3308
+ parentCandidateIds: sortedStrings(event.lineage.parentCandidateIds)
3309
+ },
3310
+ surfaces: [...event.surfaces].map((surface) => ({
3311
+ ...surface,
3312
+ artifact: { ...surface.artifact }
3313
+ })).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
3314
+ };
3315
+ if (event.kind === "task-attempted") return {
3316
+ ...event,
3317
+ artifacts,
3318
+ surfaceEvidence: [...event.surfaceEvidence].map((evidence) => ({
3319
+ ...evidence,
3320
+ evidence: sortArtifacts(evidence.evidence)
3321
+ })).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
3322
+ };
3323
+ return {
3324
+ ...event,
3325
+ artifacts
3326
+ };
3327
+ }
3328
+ function sortArtifacts(artifacts) {
3329
+ return [...artifacts].map((artifact) => ({ ...artifact })).sort((a, b) => compareStrings(artifactKey(a), artifactKey(b)));
3330
+ }
3331
+ function artifactKey(artifact) {
3332
+ return canonicalString(artifact);
3333
+ }
3334
+ function compareStrings(a, b) {
3335
+ return a < b ? -1 : a > b ? 1 : 0;
3336
+ }
3337
+ function sortedStrings(values) {
3338
+ return [...values].sort();
3339
+ }
3340
+ function assertUnique(values, label, eventId) {
3341
+ if (new Set(values).size !== values.length) throw new SearchLedgerIntegrityError(`event ${eventId} contains duplicate ${label} values`);
3342
+ }
3343
+ function formatZodError(error) {
3344
+ return error.issues.map((issue) => `${issue.path.length > 0 ? issue.path.join(".") : "<root>"}: ${issue.message}`).join("; ");
3345
+ }
3346
+ //#endregion
3347
+ //#region src/campaign/transient-failure.ts
3348
+ const BASE_TRANSIENT = /\b50[234]\b|no stream output|produced no stream|admission timed out|admission_rejected|queue_timeout|fetch failed|ECONNRESET|This operation was aborted/i;
3349
+ const TIMEOUT_PATTERN = /timeout after \d+ ?ms|cli-bridge timeout/i;
3350
+ /**
3351
+ * True when the error text describes an infrastructure hiccup that should be
3352
+ * retried rather than scored. Empty/undefined input is not transient.
3353
+ */
3354
+ function isTransientTransportFailure(message, opts = {}) {
3355
+ if (!message) return false;
3356
+ if (BASE_TRANSIENT.test(message)) return true;
3357
+ if ((opts.retryFullDurationTimeouts ?? false) && TIMEOUT_PATTERN.test(message)) return true;
3358
+ for (const p of opts.extraPatterns ?? []) if (p.test(message)) return true;
3359
+ return false;
3360
+ }
3361
+ //#endregion
3362
+ //#region src/campaign/worktree/index.ts
3363
+ /**
3364
+ * VCS-pluggable worktree adapter. One improvement = one worktree, PR-like
3365
+ * (multiple commits allowed). A code-tier proposer's `propose()` creates a
3366
+ * worktree, an agent commits the change into it, and `finalize()` returns a
3367
+ * content-addressed `CodeSurface` the measurement verifies before running.
3368
+ * On promotion the worktree becomes the PR branch.
3369
+ *
3370
+ * The interface is VCS-agnostic so a future `jj` ([jj-vcs](https://github.com/jj-vcs/jj))
3371
+ * adapter can slot in without touching proposer code. Only the git adapter
3372
+ * ships today. See `docs/design/loop-taxonomy.md`.
3373
+ */
3374
+ const MAX_GIT_OUTPUT_BYTES = 256 * 1024 * 1024;
3375
+ const FILE_HASH_CHUNK_BYTES = 1024 * 1024;
3376
+ const GIT_REPOSITORY_ENV = /* @__PURE__ */ new Set([
3377
+ "GIT_ALTERNATE_OBJECT_DIRECTORIES",
3378
+ "GIT_COMMON_DIR",
3379
+ "GIT_DIR",
3380
+ "GIT_INDEX_FILE",
3381
+ "GIT_NAMESPACE",
3382
+ "GIT_OBJECT_DIRECTORY",
3383
+ "GIT_PREFIX",
3384
+ "GIT_QUARANTINE_PATH",
3385
+ "GIT_WORK_TREE"
3386
+ ]);
3387
+ /** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
3388
+ var WorktreeAdapterError = class extends Error {
3389
+ cause;
3390
+ constructor(message, cause) {
3391
+ super(message);
3392
+ this.cause = cause;
3393
+ this.name = "WorktreeAdapterError";
3394
+ }
3395
+ };
3396
+ function defaultGit(args, cwd, overrides) {
3397
+ try {
3398
+ const env = { ...process.env };
3399
+ for (const key of Object.keys(env)) if (GIT_REPOSITORY_ENV.has(key) || key === "GIT_CONFIG" || key.startsWith("GIT_CONFIG_") || key.startsWith("GIT_ATTR_") || key === "GIT_DIFF_OPTS" || key === "GIT_EXTERNAL_DIFF") delete env[key];
3400
+ Object.assign(env, overrides);
3401
+ env.GIT_NO_REPLACE_OBJECTS = "1";
3402
+ env.LC_ALL = "C";
3403
+ env.LANG = "C";
3404
+ return execFileSync("git", args, {
3405
+ cwd,
3406
+ env,
3407
+ maxBuffer: MAX_GIT_OUTPUT_BYTES
3408
+ });
3409
+ } catch (err) {
3410
+ const stderr = err && typeof err === "object" && "stderr" in err ? String(err.stderr) : "";
3411
+ throw new WorktreeAdapterError(`git ${args.join(" ")} failed: ${stderr || String(err)}`, err);
3412
+ }
3413
+ }
3414
+ function gitBytes(git, args, cwd, env) {
3415
+ try {
3416
+ const output = git(args, cwd, env);
3417
+ return typeof output === "string" ? Buffer.from(output, "utf8") : Buffer.from(output);
3418
+ } catch (err) {
3419
+ if (err instanceof WorktreeAdapterError) throw err;
3420
+ throw new WorktreeAdapterError(`git ${args.join(" ")} failed: ${String(err)}`, err);
3421
+ }
3422
+ }
3423
+ function gitText(git, args, cwd, env) {
3424
+ return gitBytes(git, args, cwd, env).toString("utf8").trim();
3425
+ }
3426
+ function hasRegisteredWorktree(git, repoRoot, path) {
3427
+ const expected = Buffer.from(`worktree ${resolve(path)}`, "utf8");
3428
+ const records = gitBytes(git, [
3429
+ "worktree",
3430
+ "list",
3431
+ "--porcelain",
3432
+ "-z"
3433
+ ], repoRoot);
3434
+ let start = 0;
3435
+ while (start < records.length) {
3436
+ const end = records.indexOf(0, start);
3437
+ if (end < 0) throw new WorktreeAdapterError("Git worktree list output was not NUL-terminated");
3438
+ if (records.subarray(start, end).equals(expected)) return true;
3439
+ start = end + 1;
3440
+ }
3441
+ return false;
3442
+ }
3443
+ function hasLocalBranch(git, repoRoot, branch) {
3444
+ const ref = `refs/heads/${branch}`;
3445
+ return gitText(git, [
3446
+ "for-each-ref",
3447
+ "--format=%(refname)",
3448
+ "--",
3449
+ ref
3450
+ ], repoRoot).split("\n").some((candidate) => candidate === ref);
3451
+ }
3452
+ function reconcileAbsent(exists, remove) {
3453
+ try {
3454
+ if (!exists()) return void 0;
3455
+ } catch (err) {
3456
+ return err;
3457
+ }
3458
+ try {
3459
+ remove();
3460
+ return;
3461
+ } catch (removeError) {
3462
+ try {
3463
+ if (!exists()) return void 0;
3464
+ } catch (recheckError) {
3465
+ return new AggregateError([removeError, recheckError], "Removal failed and the resulting resource state could not be checked");
3466
+ }
3467
+ return removeError;
3468
+ }
3469
+ }
3470
+ function sha256(bytes) {
3471
+ return `sha256:${createHash("sha256").update(bytes).digest("hex")}`;
3472
+ }
3473
+ const PATCH_ARGS = [
3474
+ "-c",
3475
+ "core.compression=0",
3476
+ "-c",
3477
+ `core.attributesFile=${devNull}`,
3478
+ "-c",
3479
+ "core.quotePath=true",
3480
+ "-c",
3481
+ "diff.suppressBlankEmpty=false",
3482
+ "diff",
3483
+ "--unified=3",
3484
+ "--inter-hunk-context=0",
3485
+ `-O${devNull}`,
3486
+ "--binary",
3487
+ "--full-index",
3488
+ "--no-color",
3489
+ "--no-ext-diff",
3490
+ "--no-textconv",
3491
+ "--no-renames",
3492
+ "--diff-algorithm=myers",
3493
+ "--no-indent-heuristic",
3494
+ "--src-prefix=a/",
3495
+ "--dst-prefix=b/"
3496
+ ];
3497
+ const CANONICAL_GIT_ENV = {
3498
+ GIT_ATTR_GLOBAL: devNull,
3499
+ GIT_ATTR_NOSYSTEM: "1",
3500
+ GIT_CONFIG_GLOBAL: devNull,
3501
+ GIT_CONFIG_NOSYSTEM: "1",
3502
+ GIT_CONFIG_SYSTEM: devNull
3503
+ };
3504
+ /** Render the transport patch from immutable objects through fresh Git metadata.
3505
+ * Source-repository config, info attributes, templates, and caller environment
3506
+ * therefore cannot alter bytes for the same two trees. */
3507
+ function patchBytes(git, cwd, baseCommit, candidateCommit) {
3508
+ const scratch = mkdtempSync(join(tmpdir(), "agent-eval-patch-"));
3509
+ const bareRepo = join(scratch, "repo.git");
3510
+ const emptyTemplate = join(scratch, "empty-template");
3511
+ mkdirSync(emptyTemplate);
3512
+ try {
3513
+ const objectFormat = gitObjectHashAlgorithm(candidateCommit);
3514
+ const sourceObjects = realpathSync(gitText(git, [
3515
+ "rev-parse",
3516
+ "--git-path",
3517
+ "objects"
3518
+ ], cwd));
3519
+ gitText(git, [
3520
+ "init",
3521
+ "--bare",
3522
+ "--quiet",
3523
+ ...objectFormat === "sha256" ? ["--object-format=sha256"] : [],
3524
+ `--template=${emptyTemplate}`,
3525
+ bareRepo
3526
+ ], scratch, CANONICAL_GIT_ENV);
3527
+ return gitBytes(git, [
3528
+ `--git-dir=${bareRepo}`,
3529
+ ...PATCH_ARGS,
3530
+ baseCommit,
3531
+ candidateCommit,
3532
+ "--"
3533
+ ], scratch, {
3534
+ ...CANONICAL_GIT_ENV,
3535
+ GIT_ALTERNATE_OBJECT_DIRECTORIES: sourceObjects
3536
+ });
3537
+ } finally {
3538
+ rmSync(scratch, {
3539
+ recursive: true,
3540
+ force: true
3541
+ });
3542
+ }
3543
+ }
3544
+ function resolveCommit(git, cwd, ref) {
3545
+ return gitText(git, [
3546
+ "rev-parse",
3547
+ "--verify",
3548
+ `${ref}^{commit}`
3549
+ ], cwd);
3550
+ }
3551
+ function unresolvedWorktreePath(surface, worktreeDir) {
3552
+ if (isAbsolute(surface.worktreeRef)) return surface.worktreeRef;
3553
+ if (worktreeDir) return join(worktreeDir, basename(surface.worktreeRef));
3554
+ return surface.worktreeRef;
3555
+ }
3556
+ function displayGitPath(path) {
3557
+ return JSON.stringify(path);
3558
+ }
3559
+ function parseGitTreeEntries(bytes) {
3560
+ const input = Buffer.from(bytes);
3561
+ const entries = [];
3562
+ let start = 0;
3563
+ while (start < input.length) {
3564
+ const end = input.indexOf(0, start);
3565
+ if (end < 0) throw new WorktreeAdapterError("Git tree output was not NUL-terminated");
3566
+ const record = input.subarray(start, end);
3567
+ start = end + 1;
3568
+ if (record.length === 0) continue;
3569
+ const tab = record.indexOf(9);
3570
+ if (tab < 0) throw new WorktreeAdapterError("Git tree entry did not contain a path");
3571
+ const metadata = record.subarray(0, tab).toString("ascii").split(" ");
3572
+ if (metadata.length !== 3) throw new WorktreeAdapterError("Git tree entry was malformed");
3573
+ const mode = metadata[0];
3574
+ const type = metadata[1];
3575
+ const objectId = metadata[2];
3576
+ if (!mode || !type || !objectId) throw new WorktreeAdapterError("Git tree entry was malformed");
3577
+ const pathBytes = record.subarray(tab + 1);
3578
+ const path = pathBytes.toString("utf8");
3579
+ if (!Buffer.from(path, "utf8").equals(pathBytes)) throw new WorktreeAdapterError("CodeSurface paths must be valid UTF-8");
3580
+ if (type !== "blob" && type !== "commit") throw new WorktreeAdapterError(`CodeSurface contains unsupported Git object type ${type} at ${displayGitPath(path)}`);
3581
+ entries.push({
3582
+ mode,
3583
+ objectId,
3584
+ path
3585
+ });
3586
+ }
3587
+ return entries;
3588
+ }
3589
+ function parseGitPathList(bytes) {
3590
+ const input = Buffer.from(bytes);
3591
+ const paths = [];
3592
+ let start = 0;
3593
+ while (start < input.length) {
3594
+ const end = input.indexOf(0, start);
3595
+ if (end < 0) throw new WorktreeAdapterError("Git path output was not NUL-terminated");
3596
+ const pathBytes = input.subarray(start, end);
3597
+ start = end + 1;
3598
+ if (pathBytes.length === 0) continue;
3599
+ const path = pathBytes.toString("utf8");
3600
+ if (!Buffer.from(path, "utf8").equals(pathBytes)) throw new WorktreeAdapterError("CodeSurface paths must be valid UTF-8");
3601
+ paths.push(path);
3602
+ }
3603
+ return paths;
3604
+ }
3605
+ function isWithinRoot(root, candidate) {
3606
+ const rel = relative(root, candidate);
3607
+ return rel === "" || !isAbsolute(rel) && rel !== ".." && !rel.startsWith(`..${sep}`);
3608
+ }
3609
+ function assertSafeRelativePath(root, path) {
3610
+ if (path.length === 0 || isAbsolute(path)) throw new WorktreeAdapterError(`CodeSurface contains unsafe path ${displayGitPath(path)}`);
3611
+ const absolutePath = resolve(root, path);
3612
+ if (!isWithinRoot(root, absolutePath) || absolutePath === root) throw new WorktreeAdapterError(`CodeSurface path escapes its worktree: ${displayGitPath(path)}`);
3613
+ const segments = path.split(/[\\/]/u);
3614
+ let parent = root;
3615
+ for (const segment of segments.slice(0, -1)) {
3616
+ if (segment.length === 0 || segment === "." || segment === "..") throw new WorktreeAdapterError(`CodeSurface contains unsafe path ${displayGitPath(path)}`);
3617
+ parent = join(parent, segment);
3618
+ const stat = lstatSync(parent);
3619
+ if (!stat.isDirectory() || stat.isSymbolicLink()) throw new WorktreeAdapterError(`CodeSurface path traverses a non-directory or symbolic link: ${displayGitPath(path)}`);
3620
+ }
3621
+ return absolutePath;
3622
+ }
3623
+ function gitObjectHashAlgorithm(objectId) {
3624
+ if (/^[a-f0-9]{40}$/.test(objectId)) return "sha1";
3625
+ if (/^[a-f0-9]{64}$/.test(objectId)) return "sha256";
3626
+ throw new WorktreeAdapterError(`Unsupported Git object id: ${objectId}`);
3627
+ }
3628
+ function hashGitBlobBytes(bytes, objectId) {
3629
+ const hash = createHash(gitObjectHashAlgorithm(objectId));
3630
+ hash.update(`blob ${bytes.byteLength}\0`);
3631
+ hash.update(bytes);
3632
+ return hash.digest("hex");
3633
+ }
3634
+ function hashGitBlobFile(path, objectId) {
3635
+ const before = lstatSync(path);
3636
+ if (!before.isFile() || before.isSymbolicLink()) throw new WorktreeAdapterError(`CodeSurface expected a regular file at ${displayGitPath(path)}`);
3637
+ const noFollow = process.platform === "win32" ? 0 : constants.O_NOFOLLOW;
3638
+ const fd = openSync(path, constants.O_RDONLY | noFollow);
3639
+ try {
3640
+ const opened = fstatSync(fd);
3641
+ if (!opened.isFile() || opened.dev !== before.dev || opened.ino !== before.ino || opened.size !== before.size) throw new WorktreeAdapterError(`CodeSurface file changed while it was being verified: ${displayGitPath(path)}`);
3642
+ const hash = createHash(gitObjectHashAlgorithm(objectId));
3643
+ hash.update(`blob ${opened.size}\0`);
3644
+ const chunk = Buffer.allocUnsafe(FILE_HASH_CHUNK_BYTES);
3645
+ let total = 0;
3646
+ while (true) {
3647
+ const count = readSync(fd, chunk, 0, chunk.length, null);
3648
+ if (count === 0) break;
3649
+ total += count;
3650
+ hash.update(chunk.subarray(0, count));
3651
+ }
3652
+ if (total !== opened.size) throw new WorktreeAdapterError(`CodeSurface file changed while it was being verified: ${displayGitPath(path)}`);
3653
+ return {
3654
+ hash: hash.digest("hex"),
3655
+ executable: (opened.mode & 73) !== 0
3656
+ };
3657
+ } finally {
3658
+ closeSync(fd);
3659
+ }
3660
+ }
3661
+ function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
3662
+ const targetBytes = readlinkSync(linkPath, { encoding: "buffer" });
3663
+ const target = targetBytes.toString("utf8");
3664
+ if (!Buffer.from(target, "utf8").equals(targetBytes)) throw new WorktreeAdapterError(`CodeSurface symbolic link has an unsafe target at ${displayGitPath(linkPath)}`);
3665
+ if (isAbsolute(target)) throw new WorktreeAdapterError(`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`);
3666
+ if (!isWithinRoot(root, resolve(dirname(linkPath), target))) throw new WorktreeAdapterError(`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`);
3667
+ let resolvedTarget;
3668
+ try {
3669
+ resolvedTarget = realpathSync(linkPath);
3670
+ } catch (err) {
3671
+ throw new WorktreeAdapterError(`CodeSurface symbolic link target is missing or cyclic at ${displayGitPath(linkPath)}`, err);
3672
+ }
3673
+ if (!isWithinRoot(root, resolvedTarget)) throw new WorktreeAdapterError(`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`);
3674
+ const targetRelative = relative(root, resolvedTarget).split(sep).join("/");
3675
+ if (!(trackedPaths.has(targetRelative) || [...trackedPaths].some((trackedPath) => trackedPath.startsWith(`${targetRelative}/`)))) throw new WorktreeAdapterError(`CodeSurface symbolic link resolves to untracked content at ${displayGitPath(linkPath)}`);
3676
+ }
3677
+ function assertRawTreeMatchesWorktree(git, root, candidateCommit) {
3678
+ const entries = parseGitTreeEntries(gitBytes(git, [
3679
+ "ls-tree",
3680
+ "-r",
3681
+ "-z",
3682
+ "--full-tree",
3683
+ candidateCommit
3684
+ ], root));
3685
+ const trackedPaths = new Set(entries.map((entry) => entry.path));
3686
+ for (const entry of entries) {
3687
+ const absolutePath = assertSafeRelativePath(root, entry.path);
3688
+ if (entry.mode === "160000" || entry.mode === "040000") throw new WorktreeAdapterError(`CodeSurface contains a Git submodule whose executable bytes are not bound: ${displayGitPath(entry.path)}`);
3689
+ if (entry.mode === "120000") {
3690
+ if (!lstatSync(absolutePath).isSymbolicLink()) throw new WorktreeAdapterError(`CodeSurface expected a symbolic link at ${displayGitPath(entry.path)}`);
3691
+ if (hashGitBlobBytes(readlinkSync(absolutePath, { encoding: "buffer" }), entry.objectId) !== entry.objectId) throw new WorktreeAdapterError(`CodeSurface raw symbolic-link bytes differ from the candidate tree at ${displayGitPath(entry.path)}`);
3692
+ assertSymlinkTargetIsBound(root, absolutePath, trackedPaths);
3693
+ continue;
3694
+ }
3695
+ if (entry.mode !== "100644" && entry.mode !== "100755") throw new WorktreeAdapterError(`CodeSurface contains unsupported Git mode ${entry.mode} at ${displayGitPath(entry.path)}`);
3696
+ const actual = hashGitBlobFile(absolutePath, entry.objectId);
3697
+ if (actual.hash !== entry.objectId) throw new WorktreeAdapterError(`CodeSurface raw tracked bytes differ from the candidate tree because the worktree changed after finalization at ${displayGitPath(entry.path)}`);
3698
+ if (process.platform !== "win32" && actual.executable !== (entry.mode === "100755")) throw new WorktreeAdapterError(`CodeSurface executable mode differs from the candidate tree at ${displayGitPath(entry.path)}`);
3699
+ }
3700
+ }
3701
+ function verifyCodeSurfaceWithGit(surface, path, git) {
3702
+ assertCodeSurfaceIdentity(surface);
3703
+ if (!existsSync(path)) throw new WorktreeAdapterError(`CodeSurface worktree does not exist: ${path}`);
3704
+ const lexicalRoot = resolve(path);
3705
+ const canonicalRoot = realpathSync(path);
3706
+ if (lstatSync(lexicalRoot).isSymbolicLink() || process.platform !== "win32" && lexicalRoot !== canonicalRoot) throw new WorktreeAdapterError(`CodeSurface worktree locator must not contain a symbolic link: ${path}`);
3707
+ const repoRoot = gitText(git, ["rev-parse", "--show-toplevel"], path);
3708
+ const canonicalRepoRoot = realpathSync(repoRoot);
3709
+ if (canonicalRepoRoot !== canonicalRoot) throw new WorktreeAdapterError(`CodeSurface worktree locator is not the repository root: expected ${repoRoot}, got ${path}`);
3710
+ const indexFlags = gitText(git, ["ls-files", "-v"], path).split("\n").filter((line) => line.length > 0 && (line.startsWith("S ") || /^[a-z]/.test(line)));
3711
+ if (indexFlags.length > 0) throw new WorktreeAdapterError(`CodeSurface worktree uses hidden index entries that cannot be verified: ${path}\n${indexFlags.join("\n")}`);
3712
+ const extraPaths = [...parseGitPathList(gitBytes(git, [
3713
+ "ls-files",
3714
+ "--others",
3715
+ "--exclude-standard",
3716
+ "-z"
3717
+ ], path)), ...parseGitPathList(gitBytes(git, [
3718
+ "ls-files",
3719
+ "--others",
3720
+ "--ignored",
3721
+ "--exclude-standard",
3722
+ "-z"
3723
+ ], path))];
3724
+ if (extraPaths.length > 0) throw new WorktreeAdapterError(`CodeSurface worktree changed after finalization: ${path}\n${extraPaths.join("\n")}`);
3725
+ const candidateCommit = resolveCommit(git, path, "HEAD");
3726
+ if (candidateCommit !== surface.candidateCommit) throw new WorktreeAdapterError(`CodeSurface candidate commit mismatch: expected ${surface.candidateCommit}, got ${candidateCommit}`);
3727
+ const baseCommit = resolveCommit(git, path, surface.baseCommit);
3728
+ if (baseCommit !== surface.baseCommit) throw new WorktreeAdapterError(`CodeSurface base commit mismatch: expected ${surface.baseCommit}, got ${baseCommit}`);
3729
+ const baseTree = gitText(git, [
3730
+ "rev-parse",
3731
+ "--verify",
3732
+ `${surface.baseCommit}^{tree}`
3733
+ ], path);
3734
+ if (baseTree !== surface.baseTree) throw new WorktreeAdapterError(`CodeSurface base tree mismatch: expected ${surface.baseTree}, got ${baseTree}`);
3735
+ const candidateTree = gitText(git, [
3736
+ "rev-parse",
3737
+ "--verify",
3738
+ `${surface.candidateCommit}^{tree}`
3739
+ ], path);
3740
+ if (candidateTree !== surface.candidateTree) throw new WorktreeAdapterError(`CodeSurface tree mismatch: expected ${surface.candidateTree}, got ${candidateTree}`);
3741
+ try {
3742
+ gitText(git, [
3743
+ "diff-index",
3744
+ "--cached",
3745
+ "--quiet",
3746
+ "--no-ext-diff",
3747
+ "--no-textconv",
3748
+ surface.candidateCommit,
3749
+ "--"
3750
+ ], path);
3751
+ } catch (err) {
3752
+ throw new WorktreeAdapterError(`CodeSurface index differs from finalized candidate: ${path}`, err);
3753
+ }
3754
+ try {
3755
+ assertRawTreeMatchesWorktree(git, canonicalRepoRoot, surface.candidateCommit);
3756
+ } catch (err) {
3757
+ if (err instanceof WorktreeAdapterError) throw err;
3758
+ throw new WorktreeAdapterError(`CodeSurface raw tree verification failed: ${path}`, err);
3759
+ }
3760
+ if (gitText(git, [
3761
+ "merge-base",
3762
+ surface.baseCommit,
3763
+ surface.candidateCommit
3764
+ ], path) !== surface.baseCommit) throw new WorktreeAdapterError(`CodeSurface candidate ${surface.candidateCommit} does not descend from base ${surface.baseCommit}`);
3765
+ const patch = patchBytes(git, path, surface.baseCommit, surface.candidateCommit);
3766
+ const actualPatchHash = sha256(patch);
3767
+ if (actualPatchHash !== surface.patch.sha256 || patch.byteLength !== surface.patch.byteLength) throw new WorktreeAdapterError(`CodeSurface patch mismatch: expected ${surface.patch.sha256}/${surface.patch.byteLength} bytes, got ${actualPatchHash}/${patch.byteLength} bytes`);
3768
+ return {
3769
+ path,
3770
+ repoRoot: canonicalRepoRoot,
3771
+ contentHash: surfaceContentHash(surface),
3772
+ patchBytes: new Uint8Array(patch)
3773
+ };
3774
+ }
3775
+ /** Slugify a label into a branch-safe segment. */
3776
+ function slug(label) {
3777
+ return label.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 48) || "candidate";
3778
+ }
3779
+ /**
3780
+ * Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
3781
+ */
3782
+ function gitWorktreeAdapter(opts) {
3783
+ const git = opts.git ?? defaultGit;
3784
+ const worktreeDir = opts.worktreeDir ?? join(opts.repoRoot, ".worktrees");
3785
+ const branchPrefix = opts.branchPrefix ?? "improve";
3786
+ return {
3787
+ async create({ baseRef, label }) {
3788
+ const id = `${slug(label)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`;
3789
+ const branch = `${branchPrefix}/${id}`;
3790
+ const path = join(worktreeDir, id);
3791
+ const baseCommit = resolveCommit(git, opts.repoRoot, baseRef);
3792
+ const baseTree = gitText(git, [
3793
+ "rev-parse",
3794
+ "--verify",
3795
+ `${baseCommit}^{tree}`
3796
+ ], opts.repoRoot);
3797
+ gitText(git, [
3798
+ "worktree",
3799
+ "add",
3800
+ "-b",
3801
+ branch,
3802
+ path,
3803
+ baseCommit
3804
+ ], opts.repoRoot);
3805
+ return {
3806
+ path,
3807
+ branch,
3808
+ baseRef,
3809
+ baseCommit,
3810
+ baseTree
3811
+ };
3812
+ },
3813
+ async finalize(worktree, summary) {
3814
+ if (gitText(git, [
3815
+ "status",
3816
+ "--porcelain=v1",
3817
+ "--untracked-files=all"
3818
+ ], worktree.path).length > 0) {
3819
+ gitText(git, ["add", "-A"], worktree.path);
3820
+ gitText(git, [
3821
+ "commit",
3822
+ "-m",
3823
+ summary
3824
+ ], worktree.path);
3825
+ }
3826
+ const candidateCommit = resolveCommit(git, worktree.path, "HEAD");
3827
+ const candidateTree = gitText(git, [
3828
+ "rev-parse",
3829
+ "--verify",
3830
+ `${candidateCommit}^{tree}`
3831
+ ], worktree.path);
3832
+ const patch = patchBytes(git, worktree.path, worktree.baseCommit, candidateCommit);
3833
+ const surface = {
3834
+ kind: "code",
3835
+ worktreeRef: worktree.path,
3836
+ baseRef: worktree.baseRef,
3837
+ baseCommit: worktree.baseCommit,
3838
+ baseTree: worktree.baseTree,
3839
+ candidateCommit,
3840
+ candidateTree,
3841
+ patch: {
3842
+ format: "git-diff-binary",
3843
+ sha256: sha256(patch),
3844
+ byteLength: patch.byteLength
3845
+ },
3846
+ summary
3847
+ };
3848
+ verifyCodeSurfaceWithGit(surface, worktree.path, git);
3849
+ return surface;
3850
+ },
3851
+ async discard(worktree) {
3852
+ const failures = [reconcileAbsent(() => hasRegisteredWorktree(git, opts.repoRoot, worktree.path), () => gitText(git, [
3853
+ "worktree",
3854
+ "remove",
3855
+ "--force",
3856
+ "--",
3857
+ worktree.path
3858
+ ], opts.repoRoot)), reconcileAbsent(() => hasLocalBranch(git, opts.repoRoot, worktree.branch), () => gitText(git, [
3859
+ "branch",
3860
+ "-D",
3861
+ "--",
3862
+ worktree.branch
3863
+ ], opts.repoRoot))].filter((failure) => failure !== void 0);
3864
+ if (failures.length > 0) {
3865
+ const cause = failures.length === 1 ? failures[0] : new AggregateError(failures, "Multiple Git resources could not be removed");
3866
+ throw new WorktreeAdapterError(`Failed to discard worktree ${worktree.path} and branch ${worktree.branch}`, cause);
3867
+ }
3868
+ }
3869
+ };
3870
+ }
3871
+ /** Verify a finalized code surface against its current checkout. This rejects
3872
+ * dirty/ignored files, moved refs, missing Git objects, raw byte/mode
3873
+ * mismatches, external symlinks, and submodules. */
3874
+ function verifyCodeSurface(surface, worktreeDir) {
3875
+ return verifyCodeSurfaceWithGit(surface, unresolvedWorktreePath(surface, worktreeDir), defaultGit);
3876
+ }
3877
+ /** Resolve a code candidate for evaluation only after verifying its immutable
3878
+ * identity against the checkout at `worktreeRef`. */
3879
+ function resolveWorktreePath(surface, worktreeDir) {
3880
+ return verifyCodeSurface(surface, worktreeDir).path;
3881
+ }
3882
+ //#endregion
3883
+ export { planEvalFixtureRun as A, harnessAxisOf as B, rolloutArgumentDiff as C, discoverEvalFixtures as D, neutralizationGate as E, HARNESS_NATIVE_MODEL as F, parseCorrectnessResponse as G, completionVerdict as H, agentProfileHash as I, verifyCompletion as K, agentProfileId as L, buildAnalystSurfaceDispatch as M, failureModeRecallJudge as N, loadEvalFixture as O, CODING_HARNESSES as P, agentProfileModelId as R, classifyUngroundedLiterals as S, sequentialPairedGate as T, createLlmCorrectnessChecker as U, extractProducedState as V, createTokenRecallChecker as W, scoreboardSummary as _, isTransientTransportFailure as a, FsLabeledScenarioStore as b, openSearchLedger as c, selectDiscriminative as d, ProfileMatrixError as f, scoreUserStory as g, renderScoreboardMarkdown as h, verifyCodeSurface as i, analyzeCrossSurfaceInteractions as j, loadEvalFixtureScenarios as k, validateSearchLedgerEvent as l, makePlaybackDispatch as m, gitWorktreeAdapter as n, FileSearchLedger as o, runProfileMatrix as p, resolveWorktreePath as r, SEARCH_LEDGER_SCHEMA as s, WorktreeAdapterError as t, scoreDiscrimination as u, userStoryScoreboard as v, sequentialDecide as w, LabeledScenarioStoreError as x, neutralizeText as y, expandProfileAxes as z };
3884
+
3885
+ //# sourceMappingURL=campaign-CBKZvQ1H.js.map