@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
@@ -1,2098 +1,1890 @@
1
- import {
2
- fromClaudeCodeSession,
3
- fromCodexSession,
4
- fromKimiCodeSession,
5
- fromOpenCodeSession,
6
- fromPiSession,
7
- fromPigraphSession,
8
- observeCodeAgentSession,
9
- parseCodeAgentJsonl
10
- } from "../chunk-SZLVEKMJ.js";
11
- import {
12
- createHostedClient
13
- } from "../chunk-DRYIUNWY.js";
14
- import {
15
- analyzeRuns,
16
- summarizeExecution
17
- } from "../chunk-M4YBQKIJ.js";
18
- import {
19
- REFERENCE_EQUIVALENCE_INPUT_LIMITS,
20
- REFERENCE_EQUIVALENCE_JUDGE_VERSION,
21
- assertOptimizationResult,
22
- buildEvidenceVector,
23
- compareOptimizationMethods,
24
- composeGate,
25
- createReferenceEquivalenceJudge,
26
- defaultProductionGate,
27
- emitLoopProvenance,
28
- externalTextOptimizationMethod,
29
- gepaOptimizationMethod,
30
- heldOutGate,
31
- heldoutSignificance,
32
- llmJudge,
33
- loopProvenanceArgsFromResult,
34
- paretoPolicy,
35
- paretoSignificanceGate,
36
- powerPreflight,
37
- runEval,
38
- runImprovementLoop,
39
- runReferenceEquivalenceJudge,
40
- skillOptOptimizationMethod,
41
- surfaceContentHash,
42
- surfaceHash
43
- } from "../chunk-2QU3YOPR.js";
44
- import {
45
- campaignSplitDigest,
46
- createRunCostLedger,
47
- fsCampaignStorage,
48
- inMemoryCampaignStorage,
49
- resolveRunDir,
50
- runCampaign
51
- } from "../chunk-C6LXANRU.js";
52
- import {
53
- buildDefaultAnalystRegistry,
54
- createChatClient
55
- } from "../chunk-BSO5JDQH.js";
56
- import "../chunk-HHWE3POT.js";
57
- import "../chunk-WGXIEX7P.js";
58
- import {
59
- FileSystemOutcomeStore,
60
- InMemoryOutcomeStore
61
- } from "../chunk-3RF76KTD.js";
62
- import "../chunk-EG66UGL4.js";
63
- import {
64
- campaignCellExecutionEvidence,
65
- campaignCellJudgeDimensions,
66
- campaignCellTaskScore,
67
- campaignCellToRunRecord
68
- } from "../chunk-E7QXT7SX.js";
69
- import "../chunk-SFLLL76A.js";
70
- import "../chunk-TJVT4QFF.js";
71
- import "../chunk-7FO3TNPI.js";
72
- import {
73
- pairedBootstrap
74
- } from "../chunk-ZHTZ4EYI.js";
75
- import "../chunk-VCZ5FQYW.js";
76
- import "../chunk-VI2UW6B6.js";
77
- import {
78
- readTaskFailureLabels,
79
- recordAggregateMeasurements,
80
- summarizeExecutionMeasurements,
81
- summarizeTraceErrors
82
- } from "../chunk-7ZZMD7UK.js";
83
- import "../chunk-PXE2VKMX.js";
84
- import "../chunk-ZET2UAYW.js";
85
- import "../chunk-GGE4NNQT.js";
86
- import {
87
- classifyOtlpSpanRole,
88
- isOtlpModelCall
89
- } from "../chunk-P6FYH6K4.js";
90
- import "../chunk-PC4UYEBM.js";
91
- import {
92
- modelHasSnapshot,
93
- parseRunRecordSafe
94
- } from "../chunk-56TAVBOK.js";
95
- import "../chunk-MA6HLL3S.js";
96
- import "../chunk-OIUOT4QD.js";
97
- import {
98
- ValidationError
99
- } from "../chunk-ONWEPEDO.js";
100
- import {
101
- LLM_MODEL_ATTR_KEYS,
102
- SPAN_KIND_ATTR_KEYS
103
- } from "../chunk-K4DBDHLK.js";
104
- import "../chunk-PZ5AY32C.js";
105
-
106
- // src/contract/self-improve.ts
1
+ import { s as ValidationError } from "../errors-8YnH8WlF.js";
2
+ import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-BuoE80Dq.js";
3
+ import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-C-vFCSEc.js";
4
+ import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
5
+ import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
6
+ import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-CUmHkGbI.js";
7
+ import { v as pairedBootstrap } from "../statistics-CnnxdpOg.js";
8
+ import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-CQZlB3et.js";
9
+ import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-C1CavBMk.js";
10
+ import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-qipEpKvY.js";
11
+ import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
12
+ import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSession, n as fromCodexSession, o as fromPigraphSession, r as fromKimiCodeSession, s as parseCodeAgentJsonl, t as fromClaudeCodeSession } from "../code-agent-session-BjkMTQ7H.js";
13
+ import { t as createHostedClient } from "../client-CYzbdJOZ.js";
14
+ import { dirname, join } from "node:path";
15
+ import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
16
+ import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, omitTopLevelDigest } from "@tangle-network/agent-interface";
17
+ //#region src/contract/self-improve.ts
18
+ /**
19
+ * Run one complete improvement job.
20
+ *
21
+ * A caller-owned `proposer` can generate candidates across local generations.
22
+ * An external `method`, such as official GEPA or SkillOpt, owns its complete
23
+ * search and returns one candidate. Both paths remeasure the selected candidate
24
+ * against cases that candidate generation never receives.
25
+ */
26
+ /** Failed self-improvement run with an immutable receipt snapshot. */
107
27
  var SelfImproveRunError = class extends Error {
108
- cost;
109
- receipts;
110
- constructor(cause, ledger) {
111
- const original = cause instanceof Error ? cause : new Error(String(cause));
112
- super(original.message, { cause: original });
113
- this.name = "SelfImproveRunError";
114
- this.cost = ledger.summary();
115
- this.receipts = ledger.list();
116
- }
28
+ cost;
29
+ receipts;
30
+ constructor(cause, ledger) {
31
+ const original = cause instanceof Error ? cause : new Error(String(cause));
32
+ super(original.message, { cause: original });
33
+ this.name = "SelfImproveRunError";
34
+ this.cost = ledger.summary();
35
+ this.receipts = ledger.list();
36
+ }
117
37
  };
118
38
  function assertSelfImproveSearchMode(opts) {
119
- if (opts.method && opts.proposer) {
120
- throw new Error("selfImprove: method and proposer are mutually exclusive");
121
- }
122
- if (!opts.method) {
123
- if (opts.selectionScenarios !== void 0) {
124
- throw new Error("selfImprove: selectionScenarios requires method");
125
- }
126
- return;
127
- }
128
- if (typeof opts.method.name !== "string" || !opts.method.name.trim() || opts.method.name.trim() !== opts.method.name || typeof opts.method.optimize !== "function") {
129
- throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
130
- }
131
- const budget = opts.budget;
132
- if (budget?.generations !== void 0 && budget.generations !== 1) {
133
- throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
134
- }
135
- if (budget?.populationSize !== void 0 && budget.populationSize !== 1) {
136
- throw new Error(
137
- "selfImprove: method owns its candidates; budget.populationSize must be 1 when set"
138
- );
139
- }
140
- if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) {
141
- throw new Error(
142
- "selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode"
143
- );
144
- }
39
+ if (opts.method && opts.proposer) throw new Error("selfImprove: method and proposer are mutually exclusive");
40
+ if (!opts.method) {
41
+ if (opts.selectionScenarios !== void 0) throw new Error("selfImprove: selectionScenarios requires method");
42
+ return;
43
+ }
44
+ if (typeof opts.method.name !== "string" || !opts.method.name.trim() || opts.method.name.trim() !== opts.method.name || typeof opts.method.optimize !== "function") throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
45
+ const budget = opts.budget;
46
+ if (budget?.generations !== void 0 && budget.generations !== 1) throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
47
+ if (budget?.populationSize !== void 0 && budget.populationSize !== 1) throw new Error("selfImprove: method owns its candidates; budget.populationSize must be 1 when set");
48
+ if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) throw new Error("selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode");
145
49
  }
146
50
  function splitMethodPartitions(searchScenarios, explicitSelection, fraction) {
147
- if (!Number.isFinite(fraction) || fraction <= 0 || fraction >= 1) {
148
- throw new Error("selfImprove: budget.selectionFraction must be in (0, 1)");
149
- }
150
- const byId = /* @__PURE__ */ new Map();
151
- for (const scenario of searchScenarios) {
152
- if (byId.has(scenario.id)) {
153
- throw new Error(`selfImprove: duplicate scenario id '${scenario.id}'`);
154
- }
155
- byId.set(scenario.id, scenario);
156
- }
157
- if (explicitSelection) {
158
- if (explicitSelection.length === 0) {
159
- throw new Error("selfImprove: selectionScenarios must not be empty");
160
- }
161
- const selectionIds = /* @__PURE__ */ new Set();
162
- for (const scenario of explicitSelection) {
163
- if (!byId.has(scenario.id)) {
164
- throw new Error(
165
- `selfImprove: selection scenario '${scenario.id}' is absent from the non-final cases`
166
- );
167
- }
168
- if (selectionIds.has(scenario.id)) {
169
- throw new Error(`selfImprove: duplicate selection scenario id '${scenario.id}'`);
170
- }
171
- selectionIds.add(scenario.id);
172
- }
173
- const train = searchScenarios.filter((scenario) => !selectionIds.has(scenario.id));
174
- if (train.length === 0) {
175
- throw new Error("selfImprove: method train split is empty");
176
- }
177
- return {
178
- train,
179
- selection: explicitSelection.map((scenario) => byId.get(scenario.id))
180
- };
181
- }
182
- if (searchScenarios.length < 2) {
183
- throw new Error("selfImprove: method requires at least two non-final scenarios");
184
- }
185
- const sorted = [...searchScenarios].sort(
186
- (a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id)
187
- );
188
- const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
189
- return {
190
- selection: sorted.slice(0, count),
191
- train: sorted.slice(count)
192
- };
51
+ if (!Number.isFinite(fraction) || fraction <= 0 || fraction >= 1) throw new Error("selfImprove: budget.selectionFraction must be in (0, 1)");
52
+ const byId = /* @__PURE__ */ new Map();
53
+ for (const scenario of searchScenarios) {
54
+ if (byId.has(scenario.id)) throw new Error(`selfImprove: duplicate scenario id '${scenario.id}'`);
55
+ byId.set(scenario.id, scenario);
56
+ }
57
+ if (explicitSelection) {
58
+ if (explicitSelection.length === 0) throw new Error("selfImprove: selectionScenarios must not be empty");
59
+ const selectionIds = /* @__PURE__ */ new Set();
60
+ for (const scenario of explicitSelection) {
61
+ if (!byId.has(scenario.id)) throw new Error(`selfImprove: selection scenario '${scenario.id}' is absent from the non-final cases`);
62
+ if (selectionIds.has(scenario.id)) throw new Error(`selfImprove: duplicate selection scenario id '${scenario.id}'`);
63
+ selectionIds.add(scenario.id);
64
+ }
65
+ const train = searchScenarios.filter((scenario) => !selectionIds.has(scenario.id));
66
+ if (train.length === 0) throw new Error("selfImprove: method train split is empty");
67
+ return {
68
+ train,
69
+ selection: explicitSelection.map((scenario) => byId.get(scenario.id))
70
+ };
71
+ }
72
+ if (searchScenarios.length < 2) throw new Error("selfImprove: method requires at least two non-final scenarios");
73
+ const sorted = [...searchScenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
74
+ const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
75
+ return {
76
+ selection: sorted.slice(0, count),
77
+ train: sorted.slice(count)
78
+ };
193
79
  }
194
80
  function safeRunComponent(value) {
195
- return value.replace(/[^a-zA-Z0-9._-]/g, "_");
81
+ return value.replace(/[^a-zA-Z0-9._-]/g, "_");
196
82
  }
197
83
  function stableScenarioHash(value) {
198
- let hash = 2166136261 >>> 0;
199
- for (let index = 0; index < value.length; index++) {
200
- hash ^= value.charCodeAt(index);
201
- hash = Math.imul(hash, 16777619) >>> 0;
202
- }
203
- return hash;
204
- }
84
+ let hash = 2166136261;
85
+ for (let index = 0; index < value.length; index++) {
86
+ hash ^= value.charCodeAt(index);
87
+ hash = Math.imul(hash, 16777619) >>> 0;
88
+ }
89
+ return hash;
90
+ }
91
+ /**
92
+ * Deterministic train/holdout split by a stable hash of `scenario.id`,
93
+ * so the same scenario set always splits the same way across runs.
94
+ */
205
95
  function splitTrainHoldout(scenarios, fraction) {
206
- const sorted = [...scenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
207
- const nHoldout = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
208
- return {
209
- holdout: sorted.slice(0, nHoldout),
210
- train: sorted.slice(nHoldout)
211
- };
96
+ const sorted = [...scenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
97
+ const nHoldout = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
98
+ return {
99
+ holdout: sorted.slice(0, nHoldout),
100
+ train: sorted.slice(nHoldout)
101
+ };
212
102
  }
213
103
  function meanComposite(byScenario) {
214
- const perScenario = {};
215
- const values = [];
216
- for (const [id, agg] of Object.entries(byScenario)) {
217
- perScenario[id] = agg.meanComposite;
218
- values.push(agg.meanComposite);
219
- }
220
- return {
221
- compositeMean: values.length === 0 ? 0 : values.reduce((s, v) => s + v, 0) / values.length,
222
- perScenario
223
- };
224
- }
104
+ const perScenario = {};
105
+ const values = [];
106
+ for (const [id, agg] of Object.entries(byScenario)) {
107
+ perScenario[id] = agg.meanComposite;
108
+ values.push(agg.meanComposite);
109
+ }
110
+ return {
111
+ compositeMean: values.length === 0 ? 0 : values.reduce((s, v) => s + v, 0) / values.length,
112
+ perScenario
113
+ };
114
+ }
115
+ /**
116
+ * Latest search campaign measured for the winner surface; the baseline search
117
+ * campaign when the winner IS the baseline. Used by the deferred-holdout
118
+ * summary, where no holdout campaign exists to summarize.
119
+ */
225
120
  function winnerSearchCampaign(result) {
226
- for (let i = result.generations.length - 1; i >= 0; i--) {
227
- const measured = result.generations[i]?.surfaces.find(
228
- (s) => s.surfaceHash === result.winnerSurfaceHash
229
- );
230
- if (measured) return measured.campaign;
231
- }
232
- return result.baselineCampaign;
233
- }
121
+ for (let i = result.generations.length - 1; i >= 0; i--) {
122
+ const measured = result.generations[i]?.surfaces.find((s) => s.surfaceHash === result.winnerSurfaceHash);
123
+ if (measured) return measured.campaign;
124
+ }
125
+ return result.baselineCampaign;
126
+ }
127
+ /**
128
+ * One-shot self-improvement loop. See module docstring for defaults +
129
+ * extension points.
130
+ *
131
+ * @example Minimum:
132
+ *
133
+ * const result = await selfImprove({
134
+ * agent: (surface, scenario, ctx) => myAgent(surface, scenario, ctx.signal),
135
+ * scenarios,
136
+ * judge,
137
+ * baselineSurface: DEFAULT_PROMPT,
138
+ * proposer,
139
+ * })
140
+ * console.log(`lift: ${result.lift.toFixed(3)} (${result.gateDecision})`)
141
+ *
142
+ * @example Distributed (workers in three regions):
143
+ *
144
+ * await selfImprove({
145
+ * agent: httpDispatch({ resolveUrl: ({ placement }) => REGION_URLS[placement!] }),
146
+ * scenarios,
147
+ * judge,
148
+ * baselineSurface: DEFAULT_PROMPT,
149
+ * cellPlacement: ({ scenario }) => scenario.region,
150
+ * budget: { maxConcurrency: 12 },
151
+ * })
152
+ */
234
153
  async function selfImprove(opts) {
235
- const startedAt = Date.now();
236
- const requestedRunDir = opts.runDir ?? (opts.method ? `.agent-eval/runs/self-improve-${startedAt}` : `mem://selfImprove-${startedAt}`);
237
- const runDir = resolveRunDir(requestedRunDir);
238
- const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
239
- const costLedger = createRunCostLedger({
240
- storage,
241
- runDir,
242
- costCeilingUsd: opts.budget?.dollars
243
- });
244
- try {
245
- return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
246
- } catch (error) {
247
- throw new SelfImproveRunError(error, costLedger);
248
- }
154
+ const startedAt = Date.now();
155
+ const runDir = resolveRunDir(opts.runDir ?? (opts.method ? `.agent-eval/runs/self-improve-${startedAt}` : `mem://selfImprove-${startedAt}`));
156
+ const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
157
+ const costLedger = createRunCostLedger({
158
+ storage,
159
+ runDir,
160
+ costCeilingUsd: opts.budget?.dollars
161
+ });
162
+ try {
163
+ return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
164
+ } catch (error) {
165
+ throw new SelfImproveRunError(error, costLedger);
166
+ }
249
167
  }
250
168
  async function runSelfImprove(opts, costLedger, startedAt, runDir, storage) {
251
- const budget = opts.budget ?? {};
252
- assertSelfImproveSearchMode(opts);
253
- const generations = opts.method ? 1 : budget.generations ?? 3;
254
- const populationSize = opts.method ? 1 : budget.populationSize ?? 2;
255
- const maxConcurrency = budget.maxConcurrency ?? 2;
256
- const holdoutFraction = budget.holdoutFraction ?? 0.25;
257
- const holdoutMode = budget.holdout ?? "measured";
258
- const holdoutDeferred = holdoutMode === "deferred";
259
- const expectUsage = opts.expectUsage ?? "assert";
260
- const explicitHoldout = budget.holdoutScenarios;
261
- const { train, holdout } = explicitHoldout ? {
262
- train: opts.scenarios.filter((s) => !explicitHoldout.some((h) => h.id === s.id)),
263
- holdout: explicitHoldout
264
- } : holdoutDeferred ? { train: opts.scenarios, holdout: [] } : splitTrainHoldout(opts.scenarios, holdoutFraction);
265
- if (train.length === 0) {
266
- throw new Error(
267
- "selfImprove: train split is empty. Reduce holdoutFraction or pass more scenarios."
268
- );
269
- }
270
- if (holdout.length === 0 && !holdoutDeferred) {
271
- throw new Error("selfImprove: holdout split is empty. Pass more scenarios.");
272
- }
273
- if (generations > 0 && !opts.proposer && !opts.method) {
274
- throw new Error(
275
- "selfImprove: method or proposer is required when budget.generations is greater than zero"
276
- );
277
- }
278
- let optimizationResult;
279
- const methodPartitions = opts.method ? splitMethodPartitions(train, opts.selectionScenarios, budget.selectionFraction ?? 0.25) : void 0;
280
- const proposer = opts.method ? {
281
- kind: `method:${opts.method.name}`,
282
- propose: async (context) => {
283
- if (context.generation > 0) return [];
284
- const result2 = await opts.method.optimize(
285
- Object.freeze({
286
- baselineSurface: structuredClone(context.currentSurface),
287
- trainScenarios: Object.freeze(
288
- methodPartitions.train.map((scenario) => structuredClone(scenario))
289
- ),
290
- selectionScenarios: Object.freeze(
291
- methodPartitions.selection.map((scenario) => structuredClone(scenario))
292
- ),
293
- dispatchWithSurface: opts.agent,
294
- judges: Object.freeze([opts.judge]),
295
- runDir: `${runDir}/optimization/${safeRunComponent(opts.method.name)}`,
296
- seed: 42,
297
- runOptions: Object.freeze({
298
- storage,
299
- maxConcurrency,
300
- reps: budget.reps,
301
- dispatchTimeoutMs: opts.dispatchTimeoutMs,
302
- expectUsage,
303
- costCeiling: budget.dollars
304
- }),
305
- costLedger
306
- })
307
- );
308
- assertOptimizationResult(opts.method.name, result2);
309
- optimizationResult = structuredClone(result2);
310
- return [
311
- {
312
- surface: structuredClone(result2.winnerSurface),
313
- label: opts.method.name,
314
- rationale: `${opts.method.name} selected this surface without final cases.`
315
- }
316
- ];
317
- }
318
- } : opts.proposer ?? {
319
- kind: "baseline-only",
320
- propose: async () => []
321
- };
322
- const gate = opts.gate ?? defaultProductionGate({
323
- holdoutScenarios: holdout,
324
- deltaThreshold: 0.05
325
- });
326
- if (opts.onProgress) {
327
- opts.onProgress({ kind: "baseline.started", scenarios: opts.scenarios.length });
328
- }
329
- const result = await runImprovementLoop({
330
- scenarios: train,
331
- baselineSurface: opts.baselineSurface,
332
- premeasuredBaseline: opts.premeasuredBaseline,
333
- dispatchWithSurface: opts.agent,
334
- proposer,
335
- judges: [opts.judge],
336
- populationSize,
337
- maxGenerations: generations,
338
- candidateConcurrency: budget.candidateConcurrency,
339
- reps: budget.reps,
340
- maxImprovementShots: budget.maxImprovementShots,
341
- holdoutScenarios: holdout,
342
- holdout: holdoutMode,
343
- gate,
344
- neutralize: opts.neutralize,
345
- autoOnPromote: opts.autoOnPromote ?? "none",
346
- ghOwner: opts.ghOwner,
347
- ghRepo: opts.ghRepo,
348
- storage,
349
- runDir,
350
- maxConcurrency,
351
- cellPlacement: opts.cellPlacement,
352
- dispatchTimeoutMs: opts.dispatchTimeoutMs,
353
- costLedger,
354
- expectUsage,
355
- labeledStore: opts.labeledStore,
356
- captureSource: opts.captureSource,
357
- analyzeGeneration: opts.analyzeGeneration,
358
- findings: opts.findings,
359
- selectionRankKey: opts.selectionRankKey
360
- });
361
- const reportSplit = holdoutDeferred ? "search" : "holdout";
362
- const reportBaselineCampaign = holdoutDeferred ? result.baselineCampaign : result.baselineOnHoldout;
363
- const reportWinnerCampaign = holdoutDeferred ? winnerSearchCampaign(result) : result.winnerOnHoldout;
364
- const baseline = meanComposite(reportBaselineCampaign.aggregates.byScenario);
365
- const winnerStats = meanComposite(reportWinnerCampaign.aggregates.byScenario);
366
- let power;
367
- const baselineHoldoutComposites = result.baselineOnHoldout.cells.filter((cell) => !cell.error).map((cell) => {
368
- const scores = Object.values(cell.judgeScores);
369
- return scores.length === 0 ? Number.NaN : scores.reduce((sum, s) => sum + s.composite, 0) / scores.length;
370
- }).filter((v) => Number.isFinite(v));
371
- if (baselineHoldoutComposites.length >= 3) {
372
- power = powerPreflight({
373
- baselineComposites: baselineHoldoutComposites,
374
- sharedScorerChannel: true
375
- });
376
- if (opts.onProgress) {
377
- opts.onProgress({
378
- kind: "power.estimated",
379
- n: power.n,
380
- sd: power.sd,
381
- mde: power.mde,
382
- underpowered: power.underpowered
383
- });
384
- }
385
- if (power.underpowered && generations > 0) {
386
- console.warn(`[selfImprove] ${power.recommendation}`);
387
- }
388
- }
389
- if (opts.onProgress) {
390
- opts.onProgress({
391
- kind: "baseline.completed",
392
- compositeMean: baseline.compositeMean,
393
- durationMs: Date.now() - startedAt
394
- });
395
- opts.onProgress({
396
- kind: "gate.decided",
397
- decision: result.gateResult.decision,
398
- // Deferred holdout has no held-out measurement: in that mode the summary
399
- // stats are search-split numbers, and emitting their delta as `lift`
400
- // would misreport a train-split delta as a held-out one. Omit instead.
401
- ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean }
402
- });
403
- }
404
- const cost = result.cost;
405
- const totalCost = cost.totalCostUsd;
406
- const insight = await analyzeRuns({
407
- runs: [
408
- ...cellsToRunRecords(
409
- reportBaselineCampaign.cells,
410
- "baseline",
411
- runDir,
412
- opts.baselineSurface,
413
- reportSplit,
414
- opts.model
415
- ),
416
- ...cellsToRunRecords(
417
- reportWinnerCampaign.cells,
418
- "winner",
419
- runDir,
420
- result.winnerSurface,
421
- reportSplit,
422
- opts.model
423
- )
424
- ],
425
- baselineCandidateId: "baseline",
426
- candidateCandidateId: "winner"
427
- });
428
- const durationMs = Date.now() - startedAt;
429
- const { record: provenance } = await emitLoopProvenance({
430
- ...loopProvenanceArgsFromResult({
431
- runId: `${runDir}#${startedAt}`,
432
- runDir,
433
- timestamp: new Date(startedAt).toISOString(),
434
- baselineSurface: opts.baselineSurface,
435
- result,
436
- costReceipts: costLedger.list(),
437
- totalCostUsd: totalCost,
438
- totalDurationMs: durationMs
439
- }),
440
- ...optimizationResult ? {
441
- optimizationMethod: {
442
- name: opts.method.name,
443
- cost: structuredClone(optimizationResult.cost),
444
- ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
445
- ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
446
- }
447
- } : {},
448
- storage,
449
- hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
450
- });
451
- if (opts.onProvenance) opts.onProvenance(provenance);
452
- const summary = {
453
- baseline,
454
- winner: {
455
- ...winnerStats,
456
- surface: result.winnerSurface,
457
- ...result.winnerLabel ? { label: result.winnerLabel } : {},
458
- ...result.winnerRationale ? { rationale: result.winnerRationale } : {}
459
- },
460
- ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
461
- diff: result.promotedDiff,
462
- provenance,
463
- gateDecision: result.gateResult.decision,
464
- generationsExplored: result.generations.length,
465
- durationMs,
466
- totalCostUsd: totalCost,
467
- cost,
468
- receipts: costLedger.list(),
469
- ...optimizationResult ? {
470
- optimization: {
471
- name: opts.method.name,
472
- cost: structuredClone(optimizationResult.cost),
473
- ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
474
- ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
475
- }
476
- } : {},
477
- insight,
478
- ...power ? { power } : {},
479
- raw: result
480
- };
481
- if (opts.hostedTenant) {
482
- try {
483
- await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
484
- } catch (err) {
485
- const msg = err instanceof Error ? err.message : String(err);
486
- console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
487
- }
488
- }
489
- return summary;
169
+ const budget = opts.budget ?? {};
170
+ assertSelfImproveSearchMode(opts);
171
+ const generations = opts.method ? 1 : budget.generations ?? 3;
172
+ const populationSize = opts.method ? 1 : budget.populationSize ?? 2;
173
+ const maxConcurrency = budget.maxConcurrency ?? 2;
174
+ const holdoutFraction = budget.holdoutFraction ?? .25;
175
+ const holdoutMode = budget.holdout ?? "measured";
176
+ const holdoutDeferred = holdoutMode === "deferred";
177
+ const expectUsage = opts.expectUsage ?? "assert";
178
+ const explicitHoldout = budget.holdoutScenarios;
179
+ const { train, holdout } = explicitHoldout ? {
180
+ train: opts.scenarios.filter((s) => !explicitHoldout.some((h) => h.id === s.id)),
181
+ holdout: explicitHoldout
182
+ } : holdoutDeferred ? {
183
+ train: opts.scenarios,
184
+ holdout: []
185
+ } : splitTrainHoldout(opts.scenarios, holdoutFraction);
186
+ if (train.length === 0) throw new Error("selfImprove: train split is empty. Reduce holdoutFraction or pass more scenarios.");
187
+ if (holdout.length === 0 && !holdoutDeferred) throw new Error("selfImprove: holdout split is empty. Pass more scenarios.");
188
+ if (generations > 0 && !opts.proposer && !opts.method) throw new Error("selfImprove: method or proposer is required when budget.generations is greater than zero");
189
+ let optimizationResult;
190
+ const methodPartitions = opts.method ? splitMethodPartitions(train, opts.selectionScenarios, budget.selectionFraction ?? .25) : void 0;
191
+ const proposer = opts.method ? {
192
+ kind: `method:${opts.method.name}`,
193
+ propose: async (context) => {
194
+ if (context.generation > 0) return [];
195
+ const result = await opts.method.optimize(Object.freeze({
196
+ baselineSurface: structuredClone(context.currentSurface),
197
+ trainScenarios: Object.freeze(methodPartitions.train.map((scenario) => structuredClone(scenario))),
198
+ selectionScenarios: Object.freeze(methodPartitions.selection.map((scenario) => structuredClone(scenario))),
199
+ dispatchWithSurface: opts.agent,
200
+ judges: Object.freeze([opts.judge]),
201
+ runDir: `${runDir}/optimization/${safeRunComponent(opts.method.name)}`,
202
+ seed: 42,
203
+ runOptions: Object.freeze({
204
+ storage,
205
+ maxConcurrency,
206
+ reps: budget.reps,
207
+ dispatchTimeoutMs: opts.dispatchTimeoutMs,
208
+ expectUsage,
209
+ costCeiling: budget.dollars
210
+ }),
211
+ costLedger
212
+ }));
213
+ assertOptimizationResult(opts.method.name, result);
214
+ optimizationResult = structuredClone(result);
215
+ return [{
216
+ surface: structuredClone(result.winnerSurface),
217
+ label: opts.method.name,
218
+ rationale: `${opts.method.name} selected this surface without final cases.`
219
+ }];
220
+ }
221
+ } : opts.proposer ?? {
222
+ kind: "baseline-only",
223
+ propose: async () => []
224
+ };
225
+ const gate = opts.gate ?? defaultProductionGate({
226
+ holdoutScenarios: holdout,
227
+ deltaThreshold: .05
228
+ });
229
+ if (opts.onProgress) opts.onProgress({
230
+ kind: "baseline.started",
231
+ scenarios: opts.scenarios.length
232
+ });
233
+ const result = await runImprovementLoop({
234
+ scenarios: train,
235
+ baselineSurface: opts.baselineSurface,
236
+ premeasuredBaseline: opts.premeasuredBaseline,
237
+ dispatchWithSurface: opts.agent,
238
+ proposer,
239
+ judges: [opts.judge],
240
+ populationSize,
241
+ maxGenerations: generations,
242
+ candidateConcurrency: budget.candidateConcurrency,
243
+ reps: budget.reps,
244
+ maxImprovementShots: budget.maxImprovementShots,
245
+ holdoutScenarios: holdout,
246
+ holdout: holdoutMode,
247
+ gate,
248
+ neutralize: opts.neutralize,
249
+ autoOnPromote: opts.autoOnPromote ?? "none",
250
+ ghOwner: opts.ghOwner,
251
+ ghRepo: opts.ghRepo,
252
+ storage,
253
+ runDir,
254
+ maxConcurrency,
255
+ cellPlacement: opts.cellPlacement,
256
+ dispatchTimeoutMs: opts.dispatchTimeoutMs,
257
+ costLedger,
258
+ expectUsage,
259
+ labeledStore: opts.labeledStore,
260
+ captureSource: opts.captureSource,
261
+ analyzeGeneration: opts.analyzeGeneration,
262
+ findings: opts.findings,
263
+ selectionRankKey: opts.selectionRankKey
264
+ });
265
+ const reportSplit = holdoutDeferred ? "search" : "holdout";
266
+ const reportBaselineCampaign = holdoutDeferred ? result.baselineCampaign : result.baselineOnHoldout;
267
+ const reportWinnerCampaign = holdoutDeferred ? winnerSearchCampaign(result) : result.winnerOnHoldout;
268
+ const baseline = meanComposite(reportBaselineCampaign.aggregates.byScenario);
269
+ const winnerStats = meanComposite(reportWinnerCampaign.aggregates.byScenario);
270
+ let power;
271
+ const baselineHoldoutComposites = result.baselineOnHoldout.cells.filter((cell) => !cell.error).map((cell) => {
272
+ const scores = Object.values(cell.judgeScores);
273
+ return scores.length === 0 ? NaN : scores.reduce((sum, s) => sum + s.composite, 0) / scores.length;
274
+ }).filter((v) => Number.isFinite(v));
275
+ if (baselineHoldoutComposites.length >= 3) {
276
+ power = powerPreflight({
277
+ baselineComposites: baselineHoldoutComposites,
278
+ sharedScorerChannel: true
279
+ });
280
+ if (opts.onProgress) opts.onProgress({
281
+ kind: "power.estimated",
282
+ n: power.n,
283
+ sd: power.sd,
284
+ mde: power.mde,
285
+ underpowered: power.underpowered
286
+ });
287
+ if (power.underpowered && generations > 0) console.warn(`[selfImprove] ${power.recommendation}`);
288
+ }
289
+ if (opts.onProgress) {
290
+ opts.onProgress({
291
+ kind: "baseline.completed",
292
+ compositeMean: baseline.compositeMean,
293
+ durationMs: Date.now() - startedAt
294
+ });
295
+ opts.onProgress({
296
+ kind: "gate.decided",
297
+ decision: result.gateResult.decision,
298
+ ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean }
299
+ });
300
+ }
301
+ const cost = result.cost;
302
+ const totalCost = cost.totalCostUsd;
303
+ const insight = await analyzeRuns({
304
+ runs: [...cellsToRunRecords(reportBaselineCampaign.cells, "baseline", runDir, opts.baselineSurface, reportSplit, opts.model), ...cellsToRunRecords(reportWinnerCampaign.cells, "winner", runDir, result.winnerSurface, reportSplit, opts.model)],
305
+ baselineCandidateId: "baseline",
306
+ candidateCandidateId: "winner"
307
+ });
308
+ const durationMs = Date.now() - startedAt;
309
+ const { record: provenance } = await emitLoopProvenance({
310
+ ...loopProvenanceArgsFromResult({
311
+ runId: `${runDir}#${startedAt}`,
312
+ runDir,
313
+ timestamp: new Date(startedAt).toISOString(),
314
+ baselineSurface: opts.baselineSurface,
315
+ result,
316
+ costReceipts: costLedger.list(),
317
+ totalCostUsd: totalCost,
318
+ totalDurationMs: durationMs
319
+ }),
320
+ ...optimizationResult ? { optimizationMethod: {
321
+ name: opts.method.name,
322
+ cost: structuredClone(optimizationResult.cost),
323
+ ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
324
+ ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
325
+ } } : {},
326
+ storage,
327
+ hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
328
+ });
329
+ if (opts.onProvenance) opts.onProvenance(provenance);
330
+ const summary = {
331
+ baseline,
332
+ winner: {
333
+ ...winnerStats,
334
+ surface: result.winnerSurface,
335
+ ...result.winnerLabel ? { label: result.winnerLabel } : {},
336
+ ...result.winnerRationale ? { rationale: result.winnerRationale } : {}
337
+ },
338
+ ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
339
+ diff: result.promotedDiff,
340
+ provenance,
341
+ gateDecision: result.gateResult.decision,
342
+ generationsExplored: result.generations.length,
343
+ durationMs,
344
+ totalCostUsd: totalCost,
345
+ cost,
346
+ receipts: costLedger.list(),
347
+ ...optimizationResult ? { optimization: {
348
+ name: opts.method.name,
349
+ cost: structuredClone(optimizationResult.cost),
350
+ ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
351
+ ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
352
+ } } : {},
353
+ insight,
354
+ ...power ? { power } : {},
355
+ raw: result
356
+ };
357
+ if (opts.hostedTenant) try {
358
+ await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
359
+ } catch (err) {
360
+ const msg = err instanceof Error ? err.message : String(err);
361
+ console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
362
+ }
363
+ return summary;
490
364
  }
491
365
  async function shipEvalRunToHosted(tenant, opts, summary, raw, runDir) {
492
- const client = createHostedClient(tenant);
493
- function snapshotFromCampaign(index, surface, campaign, durationMs) {
494
- const cells = campaign.cells.map((cell) => {
495
- const execution = campaignCellExecutionEvidence(cell);
496
- return {
497
- scenarioId: cell.scenarioId,
498
- rep: cell.rep,
499
- compositeMean: campaignCellTaskScore(cell) ?? null,
500
- dimensions: campaignCellJudgeDimensions(cell),
501
- terminalOutcome: execution.terminalOutcome,
502
- executionErrorCount: execution.executionErrorCount ?? null,
503
- errorMessage: cell.error ?? void 0
504
- };
505
- });
506
- const scoredCells = cells.flatMap(
507
- (cell) => cell.compositeMean === null ? [] : [cell.compositeMean]
508
- );
509
- const compositeMean = scoredCells.length === 0 ? null : scoredCells.reduce((sum, score) => sum + score, 0) / scoredCells.length;
510
- return {
511
- index,
512
- surfaceHash: surfaceHash(surface),
513
- surface,
514
- cells,
515
- compositeMean,
516
- costUsd: campaign.aggregates.cost.totalCostUsd,
517
- durationMs
518
- };
519
- }
520
- const generations = [];
521
- generations.push(snapshotFromCampaign(0, opts.baselineSurface, raw.baselineCampaign, 0));
522
- for (const gen of raw.generations) {
523
- const winner = gen.surfaces.reduce(
524
- (best, s) => s.campaign.aggregates.cellsExecuted > 0 && (best === void 0 || averageComposite(s.campaign) > averageComposite(best.campaign)) ? s : best,
525
- gen.surfaces[0]
526
- );
527
- if (!winner) continue;
528
- generations.push(
529
- snapshotFromCampaign(gen.record.generationIndex + 1, winner.surface, winner.campaign, 0)
530
- );
531
- }
532
- const event = {
533
- runId: `${runDir}#${Date.now()}`,
534
- runDir,
535
- timestamp: (/* @__PURE__ */ new Date()).toISOString(),
536
- status: "finished",
537
- labels: opts.hostedLabels ?? {},
538
- baseline: generations[0],
539
- generations,
540
- gateDecision: summary.gateDecision,
541
- holdoutLift: summary.lift,
542
- totalCostUsd: summary.totalCostUsd,
543
- totalDurationMs: summary.durationMs,
544
- insightReport: summary.insight
545
- };
546
- await client.ingestEvalRun(event);
366
+ const client = createHostedClient(tenant);
367
+ function snapshotFromCampaign(index, surface, campaign, durationMs) {
368
+ const cells = campaign.cells.map((cell) => {
369
+ const execution = campaignCellExecutionEvidence(cell);
370
+ return {
371
+ scenarioId: cell.scenarioId,
372
+ rep: cell.rep,
373
+ compositeMean: campaignCellTaskScore(cell) ?? null,
374
+ dimensions: campaignCellJudgeDimensions(cell),
375
+ terminalOutcome: execution.terminalOutcome,
376
+ executionErrorCount: execution.executionErrorCount ?? null,
377
+ errorMessage: cell.error ?? void 0
378
+ };
379
+ });
380
+ const scoredCells = cells.flatMap((cell) => cell.compositeMean === null ? [] : [cell.compositeMean]);
381
+ const compositeMean = scoredCells.length === 0 ? null : scoredCells.reduce((sum, score) => sum + score, 0) / scoredCells.length;
382
+ return {
383
+ index,
384
+ surfaceHash: surfaceHash(surface),
385
+ surface,
386
+ cells,
387
+ compositeMean,
388
+ costUsd: campaign.aggregates.cost.totalCostUsd,
389
+ durationMs
390
+ };
391
+ }
392
+ const generations = [];
393
+ generations.push(snapshotFromCampaign(0, opts.baselineSurface, raw.baselineCampaign, 0));
394
+ for (const gen of raw.generations) {
395
+ const winner = gen.surfaces.reduce((best, s) => s.campaign.aggregates.cellsExecuted > 0 && (best === void 0 || averageComposite(s.campaign) > averageComposite(best.campaign)) ? s : best, gen.surfaces[0]);
396
+ if (!winner) continue;
397
+ generations.push(snapshotFromCampaign(gen.record.generationIndex + 1, winner.surface, winner.campaign, 0));
398
+ }
399
+ const event = {
400
+ runId: `${runDir}#${Date.now()}`,
401
+ runDir,
402
+ timestamp: (/* @__PURE__ */ new Date()).toISOString(),
403
+ status: "finished",
404
+ labels: opts.hostedLabels ?? {},
405
+ baseline: generations[0],
406
+ generations,
407
+ gateDecision: summary.gateDecision,
408
+ holdoutLift: summary.lift,
409
+ totalCostUsd: summary.totalCostUsd,
410
+ totalDurationMs: summary.durationMs,
411
+ insightReport: summary.insight
412
+ };
413
+ await client.ingestEvalRun(event);
547
414
  }
548
415
  function averageComposite(campaign) {
549
- const aggs = Object.values(campaign.aggregates.byScenario);
550
- return aggs.length === 0 ? 0 : aggs.reduce((s, a) => s + a.meanComposite, 0) / aggs.length;
416
+ const aggs = Object.values(campaign.aggregates.byScenario);
417
+ return aggs.length === 0 ? 0 : aggs.reduce((s, a) => s + a.meanComposite, 0) / aggs.length;
551
418
  }
552
419
  function hashString(s) {
553
- let h = 2166136261 >>> 0;
554
- for (let i = 0; i < s.length; i++) {
555
- h ^= s.charCodeAt(i);
556
- h = Math.imul(h, 16777619) >>> 0;
557
- }
558
- return h.toString(16).padStart(8, "0");
559
- }
420
+ let h = 2166136261;
421
+ for (let i = 0; i < s.length; i++) {
422
+ h ^= s.charCodeAt(i);
423
+ h = Math.imul(h, 16777619) >>> 0;
424
+ }
425
+ return h.toString(16).padStart(8, "0");
426
+ }
427
+ /**
428
+ * Adapt campaign cells into the `RunRecord` shape `analyzeRuns()` consumes.
429
+ * Each cell becomes one run; `candidateId` is the caller-supplied label so
430
+ * baseline + winner pair cleanly on `(experimentId, scenarioId, seed)`.
431
+ *
432
+ * `promptHash` is the REAL sha256 content hash of the surface this cell ran
433
+ * (baseline vs winner are byte-distinguishable + byte-identical-verifiable);
434
+ * `configHash` is the sha256 of the candidate label so the two candidates'
435
+ * config rows differ. Both were previously the literal `'sha256:cell'`, which
436
+ * made baseline and winner indistinguishable in every downstream record.
437
+ */
560
438
  function cellsToRunRecords(cells, candidateId, runId, surface, splitTag, fallbackModel) {
561
- const promptHash = surfaceContentHash(surface);
562
- const configHash = surfaceContentHash(candidateId);
563
- return cells.map((cell) => {
564
- const model = cell.resolvedModel ?? fallbackModel;
565
- if (!model) {
566
- throw new ValidationError(
567
- `selfImprove.model is required when cell ${cell.cellId} has no paid-call model receipt`
568
- );
569
- }
570
- if (!modelHasSnapshot(model)) {
571
- throw new ValidationError(
572
- `selfImprove model "${model}" lacks a snapshot version for cell ${cell.cellId}`
573
- );
574
- }
575
- return campaignCellToRunRecord(cell, {
576
- runId: `${runId}::${candidateId}::${cell.cellId}`,
577
- experimentId: runId,
578
- candidateId,
579
- // scenarioId is explicit; seed keeps repeated runs distinct.
580
- seed: cell.rep * 1e6 + hashString(cell.scenarioId).slice(0, 6).split("").reduce((a, c) => a * 31 + c.charCodeAt(0) >>> 0, 0),
581
- model,
582
- promptHash,
583
- configHash,
584
- commitSha: "cell",
585
- splitTag
586
- });
587
- });
588
- }
589
-
590
- // src/contract/define-agent-eval.ts
439
+ const promptHash = surfaceContentHash(surface);
440
+ const configHash = surfaceContentHash(candidateId);
441
+ return cells.map((cell) => {
442
+ const model = cell.resolvedModel ?? fallbackModel;
443
+ if (!model) throw new ValidationError(`selfImprove.model is required when cell ${cell.cellId} has no paid-call model receipt`);
444
+ if (!modelHasSnapshot(model)) throw new ValidationError(`selfImprove model "${model}" lacks a snapshot version for cell ${cell.cellId}`);
445
+ return campaignCellToRunRecord(cell, {
446
+ runId: `${runId}::${candidateId}::${cell.cellId}`,
447
+ experimentId: runId,
448
+ candidateId,
449
+ seed: cell.rep * 1e6 + hashString(cell.scenarioId).slice(0, 6).split("").reduce((a, c) => a * 31 + c.charCodeAt(0) >>> 0, 0),
450
+ model,
451
+ promptHash,
452
+ configHash,
453
+ commitSha: "cell",
454
+ splitTag
455
+ });
456
+ });
457
+ }
458
+ //#endregion
459
+ //#region src/contract/define-agent-eval.ts
460
+ /**
461
+ * Define an agent eval once, then either score a surface with `evaluate()` or
462
+ * run the closed loop with `improve()`.
463
+ *
464
+ * This is a DX wrapper only: it delegates to `runEval()` and `selfImprove()` and
465
+ * returns their native result shapes.
466
+ */
591
467
  function defineAgentEval(defaults) {
592
- const defaultEvaluateOptions = evaluateDefaults(defaults);
593
- return {
594
- scenarios: defaults.scenarios,
595
- baselineSurface: defaults.baselineSurface,
596
- async evaluate(opts = {}) {
597
- const { agent, judge, judges, runDir, scenarios, surface, ...campaignOpts } = opts;
598
- const selectedAgent = agent ?? defaults.agent;
599
- const selectedSurface = surface ?? defaults.baselineSurface;
600
- const selectedRunDir = runDir ?? defaults.runDir ?? `mem://defineAgentEval-${Date.now()}`;
601
- const selectedStorage = campaignOpts.storage ?? defaultEvaluateOptions.storage ?? (selectedRunDir.startsWith("mem://") ? inMemoryCampaignStorage() : void 0);
602
- const evalOptions = {
603
- ...defaultEvaluateOptions,
604
- ...campaignOpts,
605
- ...selectedStorage ? { storage: selectedStorage } : {},
606
- runDir: selectedRunDir,
607
- scenarios: scenarios ?? defaults.scenarios,
608
- dispatch: (scenario, ctx) => selectedAgent(selectedSurface, scenario, ctx),
609
- judges: evaluateJudges(judges, judge ?? defaults.judge)
610
- };
611
- if (evalOptions.reps !== void 0)
612
- evalOptions.reps = requirePositiveInteger(evalOptions.reps, "reps");
613
- return runEval(evalOptions);
614
- },
615
- async improve(opts = {}) {
616
- const {
617
- budget: budgetOverride,
618
- hostedTenant: hostedTenantOverride,
619
- ...topLevelOverrides
620
- } = opts;
621
- const merged = mergeDefined(defaults, topLevelOverrides);
622
- const budget = mergeBudget(defaults.budget, budgetOverride);
623
- const hostedTenant = mergeHostedTenant(defaults.hostedTenant, hostedTenantOverride);
624
- return selfImprove({
625
- ...merged,
626
- ...budget ? { budget } : {},
627
- ...hostedTenant ? { hostedTenant } : {}
628
- });
629
- }
630
- };
468
+ const defaultEvaluateOptions = evaluateDefaults(defaults);
469
+ return {
470
+ scenarios: defaults.scenarios,
471
+ baselineSurface: defaults.baselineSurface,
472
+ async evaluate(opts = {}) {
473
+ const { agent, judge, judges, runDir, scenarios, surface, ...campaignOpts } = opts;
474
+ const selectedAgent = agent ?? defaults.agent;
475
+ const selectedSurface = surface ?? defaults.baselineSurface;
476
+ const selectedRunDir = runDir ?? defaults.runDir ?? `mem://defineAgentEval-${Date.now()}`;
477
+ const selectedStorage = campaignOpts.storage ?? defaultEvaluateOptions.storage ?? (selectedRunDir.startsWith("mem://") ? inMemoryCampaignStorage() : void 0);
478
+ const evalOptions = {
479
+ ...defaultEvaluateOptions,
480
+ ...campaignOpts,
481
+ ...selectedStorage ? { storage: selectedStorage } : {},
482
+ runDir: selectedRunDir,
483
+ scenarios: scenarios ?? defaults.scenarios,
484
+ dispatch: (scenario, ctx) => selectedAgent(selectedSurface, scenario, ctx),
485
+ judges: evaluateJudges(judges, judge ?? defaults.judge)
486
+ };
487
+ if (evalOptions.reps !== void 0) evalOptions.reps = requirePositiveInteger(evalOptions.reps, "reps");
488
+ return runEval(evalOptions);
489
+ },
490
+ async improve(opts = {}) {
491
+ const { budget: budgetOverride, hostedTenant: hostedTenantOverride, ...topLevelOverrides } = opts;
492
+ const merged = mergeDefined(defaults, topLevelOverrides);
493
+ const budget = mergeBudget(defaults.budget, budgetOverride);
494
+ const hostedTenant = mergeHostedTenant(defaults.hostedTenant, hostedTenantOverride);
495
+ return selfImprove({
496
+ ...merged,
497
+ ...budget ? { budget } : {},
498
+ ...hostedTenant ? { hostedTenant } : {}
499
+ });
500
+ }
501
+ };
631
502
  }
632
503
  function evaluateDefaults(defaults) {
633
- const out = {};
634
- if (defaults.storage) out.storage = defaults.storage;
635
- if (defaults.labeledStore) out.labeledStore = defaults.labeledStore;
636
- if (defaults.captureSource) out.captureSource = defaults.captureSource;
637
- if (defaults.cellPlacement) out.cellPlacement = defaults.cellPlacement;
638
- if (defaults.expectUsage) out.expectUsage = defaults.expectUsage;
639
- if (defaults.budget?.dollars !== void 0) out.costCeiling = defaults.budget.dollars;
640
- if (defaults.budget?.maxConcurrency !== void 0)
641
- out.maxConcurrency = defaults.budget.maxConcurrency;
642
- if (defaults.budget?.reps !== void 0)
643
- out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
644
- return out;
504
+ const out = {};
505
+ if (defaults.storage) out.storage = defaults.storage;
506
+ if (defaults.labeledStore) out.labeledStore = defaults.labeledStore;
507
+ if (defaults.captureSource) out.captureSource = defaults.captureSource;
508
+ if (defaults.cellPlacement) out.cellPlacement = defaults.cellPlacement;
509
+ if (defaults.expectUsage) out.expectUsage = defaults.expectUsage;
510
+ if (defaults.budget?.dollars !== void 0) out.costCeiling = defaults.budget.dollars;
511
+ if (defaults.budget?.maxConcurrency !== void 0) out.maxConcurrency = defaults.budget.maxConcurrency;
512
+ if (defaults.budget?.reps !== void 0) out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
513
+ return out;
645
514
  }
646
515
  function mergeBudget(defaults, overrides) {
647
- const merged = mergeOptionalObject(defaults, overrides);
648
- if (merged?.reps !== void 0) merged.reps = requirePositiveInteger(merged.reps, "budget.reps");
649
- return merged;
516
+ const merged = mergeOptionalObject(defaults, overrides);
517
+ if (merged?.reps !== void 0) merged.reps = requirePositiveInteger(merged.reps, "budget.reps");
518
+ return merged;
650
519
  }
651
520
  function mergeHostedTenant(defaults, overrides) {
652
- const merged = mergeOptionalObject(defaults, overrides);
653
- if (!merged) return void 0;
654
- if (!merged.endpoint?.trim() || !merged.apiKey?.trim() || !merged.tenantId?.trim()) {
655
- throw new Error(
656
- "defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides"
657
- );
658
- }
659
- return merged;
521
+ const merged = mergeOptionalObject(defaults, overrides);
522
+ if (!merged) return void 0;
523
+ if (!merged.endpoint?.trim() || !merged.apiKey?.trim() || !merged.tenantId?.trim()) throw new Error("defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides");
524
+ return merged;
660
525
  }
661
526
  function mergeDefined(defaults, overrides) {
662
- if (!overrides) return defaults;
663
- const merged = { ...defaults };
664
- for (const [key, value] of Object.entries(overrides)) {
665
- if (value !== void 0) merged[key] = value;
666
- }
667
- return merged;
527
+ if (!overrides) return defaults;
528
+ const merged = { ...defaults };
529
+ for (const [key, value] of Object.entries(overrides)) if (value !== void 0) merged[key] = value;
530
+ return merged;
668
531
  }
669
532
  function mergeOptionalObject(defaults, overrides) {
670
- if (!defaults && !overrides) return void 0;
671
- return mergeDefined(defaults ?? {}, overrides);
533
+ if (!defaults && !overrides) return void 0;
534
+ return mergeDefined(defaults ?? {}, overrides);
672
535
  }
673
536
  function evaluateJudges(judges, defaultJudge) {
674
- if (judges !== void 0) {
675
- if (judges.length === 0) {
676
- throw new Error("defineAgentEval.evaluate: judges must not be empty");
677
- }
678
- return judges;
679
- }
680
- return [defaultJudge];
537
+ if (judges !== void 0) {
538
+ if (judges.length === 0) throw new Error("defineAgentEval.evaluate: judges must not be empty");
539
+ return judges;
540
+ }
541
+ return [defaultJudge];
681
542
  }
682
543
  function requirePositiveInteger(value, field) {
683
- if (!Number.isInteger(value) || value < 1) {
684
- throw new Error(`defineAgentEval: ${field} must be a positive integer`);
685
- }
686
- return value;
544
+ if (!Number.isInteger(value) || value < 1) throw new Error(`defineAgentEval: ${field} must be a positive integer`);
545
+ return value;
687
546
  }
688
-
689
- // src/contract/measured-comparison.ts
690
- import {
691
- agentCandidateBenchmarkSuiteSchema,
692
- agentCandidateBenchmarkTaskSchema,
693
- agentCandidateBundleSchema,
694
- agentCandidateEvaluationPolicySchema,
695
- agentCandidateExperimentSchema,
696
- agentImprovementMeasuredComparisonSchema,
697
- candidateExecutionEvidenceSchema,
698
- canonicalCandidateDigest,
699
- omitTopLevelDigest
700
- } from "@tangle-network/agent-interface";
547
+ //#endregion
548
+ //#region src/contract/measured-comparison.ts
549
+ /** Content-address one task before any measured execution can see it. */
701
550
  function sealCandidateBenchmarkTask(material) {
702
- return agentCandidateBenchmarkTaskSchema.parse({
703
- ...material,
704
- digest: canonicalCandidateDigest(material)
705
- });
551
+ return agentCandidateBenchmarkTaskSchema.parse({
552
+ ...material,
553
+ digest: canonicalCandidateDigest(material)
554
+ });
706
555
  }
556
+ /** Freeze task order, repetitions, and every seed before either arm runs. */
707
557
  function sealCandidateBenchmarkSuite(options) {
708
- for (const task of options.tasks) verifyCandidateBenchmarkTask(task);
709
- const material = {
710
- kind: "agent-candidate-benchmark-suite",
711
- digestAlgorithm: "rfc8785-sha256",
712
- taskDigests: options.tasks.map((task) => task.digest),
713
- reps: options.reps,
714
- seeds: options.seeds
715
- };
716
- const suite = agentCandidateBenchmarkSuiteSchema.parse({
717
- ...material,
718
- digest: canonicalCandidateDigest(material)
719
- });
720
- return { suite, tasks: options.tasks };
721
- }
558
+ for (const task of options.tasks) verifyCandidateBenchmarkTask(task);
559
+ const material = {
560
+ kind: "agent-candidate-benchmark-suite",
561
+ digestAlgorithm: "rfc8785-sha256",
562
+ taskDigests: options.tasks.map((task) => task.digest),
563
+ reps: options.reps,
564
+ seeds: options.seeds
565
+ };
566
+ return {
567
+ suite: agentCandidateBenchmarkSuiteSchema.parse({
568
+ ...material,
569
+ digest: canonicalCandidateDigest(material)
570
+ }),
571
+ tasks: options.tasks
572
+ };
573
+ }
574
+ /** Freeze both complete agent states and their exact held-out work. */
722
575
  function sealCandidateExperiment(material) {
723
- const parsed = agentCandidateExperimentSchema.parse({
724
- ...material,
725
- digest: canonicalCandidateDigest(material)
726
- });
727
- return verifyCandidateExperiment(parsed);
576
+ return verifyCandidateExperiment(agentCandidateExperimentSchema.parse({
577
+ ...material,
578
+ digest: canonicalCandidateDigest(material)
579
+ }));
728
580
  }
729
581
  function verifyCandidateExperiment(input) {
730
- const experiment = agentCandidateExperimentSchema.parse(input);
731
- verifySelfAddressed(experiment, "candidate experiment");
732
- verifyBundle(experiment.baseline, "baseline bundle");
733
- verifyBundle(experiment.candidate, "candidate bundle");
734
- if (experiment.baseline.digest === experiment.candidate.digest) {
735
- throw new Error("candidate experiment baseline and candidate bundles are identical");
736
- }
737
- verifyCandidateBenchmarkSuiteInputs(experiment.benchmark);
738
- return experiment;
739
- }
582
+ const experiment = agentCandidateExperimentSchema.parse(input);
583
+ verifySelfAddressed(experiment, "candidate experiment");
584
+ verifyBundle(experiment.baseline, "baseline bundle");
585
+ verifyBundle(experiment.candidate, "candidate bundle");
586
+ if (experiment.baseline.digest === experiment.candidate.digest) throw new Error("candidate experiment baseline and candidate bundles are identical");
587
+ verifyCandidateBenchmarkSuiteInputs(experiment.benchmark);
588
+ return experiment;
589
+ }
590
+ /** Execute each signed cell for both arms. The callback is Runtime's one executor. */
740
591
  async function runCandidateExperiment(options) {
741
- const experiment = verifyCandidateExperiment(options.experiment);
742
- const { suite, tasks } = experiment.benchmark;
743
- const maxConcurrency = options.maxConcurrency ?? 2;
744
- if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) {
745
- throw new Error("candidate experiment maxConcurrency must be a positive integer");
746
- }
747
- const measurements = new Array(
748
- suite.taskDigests.length * suite.reps
749
- );
750
- let nextIndex = 0;
751
- const lanes = Array.from({ length: Math.min(maxConcurrency, measurements.length) }, async () => {
752
- while (true) {
753
- if (options.signal?.aborted) throw abortError(options.signal);
754
- const index = nextIndex;
755
- nextIndex += 1;
756
- if (index >= measurements.length) return;
757
- const taskIndex = Math.floor(index / suite.reps);
758
- const repetition = index % suite.reps;
759
- const task = tasks[taskIndex];
760
- const seed = suite.seeds[index];
761
- if (!task || seed === void 0) {
762
- throw new Error(`candidate experiment cell ${index} has no signed task or seed`);
763
- }
764
- const benchmarkCell = {
765
- suiteDigest: suite.digest,
766
- taskIndex,
767
- repetition
768
- };
769
- const [baseline, candidate] = await Promise.all([
770
- options.execute({
771
- experiment,
772
- arm: "baseline",
773
- bundle: experiment.baseline,
774
- task,
775
- benchmarkCell,
776
- seed,
777
- ...options.signal ? { signal: options.signal } : {}
778
- }),
779
- options.execute({
780
- experiment,
781
- arm: "candidate",
782
- bundle: experiment.candidate,
783
- task,
784
- benchmarkCell,
785
- seed,
786
- ...options.signal ? { signal: options.signal } : {}
787
- })
788
- ]);
789
- const measurement = { baseline, candidate };
790
- verifyMeasurement(experiment, measurement, index);
791
- measurements[index] = measurement;
792
- }
793
- });
794
- await Promise.all(lanes);
795
- return measurements;
796
- }
592
+ const experiment = verifyCandidateExperiment(options.experiment);
593
+ const { suite, tasks } = experiment.benchmark;
594
+ const maxConcurrency = options.maxConcurrency ?? 2;
595
+ if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) throw new Error("candidate experiment maxConcurrency must be a positive integer");
596
+ const measurements = new Array(suite.taskDigests.length * suite.reps);
597
+ let nextIndex = 0;
598
+ const lanes = Array.from({ length: Math.min(maxConcurrency, measurements.length) }, async () => {
599
+ while (true) {
600
+ if (options.signal?.aborted) throw abortError(options.signal);
601
+ const index = nextIndex;
602
+ nextIndex += 1;
603
+ if (index >= measurements.length) return;
604
+ const taskIndex = Math.floor(index / suite.reps);
605
+ const repetition = index % suite.reps;
606
+ const task = tasks[taskIndex];
607
+ const seed = suite.seeds[index];
608
+ if (!task || seed === void 0) throw new Error(`candidate experiment cell ${index} has no signed task or seed`);
609
+ const benchmarkCell = {
610
+ suiteDigest: suite.digest,
611
+ taskIndex,
612
+ repetition
613
+ };
614
+ const [baseline, candidate] = await Promise.all([options.execute({
615
+ experiment,
616
+ arm: "baseline",
617
+ bundle: experiment.baseline,
618
+ task,
619
+ benchmarkCell,
620
+ seed,
621
+ ...options.signal ? { signal: options.signal } : {}
622
+ }), options.execute({
623
+ experiment,
624
+ arm: "candidate",
625
+ bundle: experiment.candidate,
626
+ task,
627
+ benchmarkCell,
628
+ seed,
629
+ ...options.signal ? { signal: options.signal } : {}
630
+ })]);
631
+ const measurement = {
632
+ baseline,
633
+ candidate
634
+ };
635
+ verifyMeasurement(experiment, measurement, index);
636
+ measurements[index] = measurement;
637
+ }
638
+ });
639
+ await Promise.all(lanes);
640
+ return measurements;
641
+ }
642
+ /**
643
+ * Calculate the shared paired decision from any complete receipt shape.
644
+ *
645
+ * Callers still own sealing their tasks, verifying each receipt against its
646
+ * expected arm and state, and proving every expected cell exists. This function
647
+ * only validates the projected measurements and derives their shared decision.
648
+ */
797
649
  function evaluatePairedMeasurements(options) {
798
- if (options.measurements.length === 0) {
799
- throw new Error("paired measurement evaluation requires at least one paired cell");
800
- }
801
- const additionalCostUsd = options.additionalCostUsd ?? 0;
802
- if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) {
803
- throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
804
- }
805
- if (typeof options.sharedScorerChannel !== "boolean") {
806
- throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
807
- }
808
- const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
809
- const measurements = options.measurements.map(
810
- (measurement, index) => projectPairedMeasurement(measurement, index, options.adapter)
811
- );
812
- const cellIds2 = measurements.map((measurement) => measurement.cellId);
813
- if (new Set(cellIds2).size !== cellIds2.length) {
814
- throw new Error("paired measurement evaluation cell ids must be unique");
815
- }
816
- const dimensions = sharedProjectedDimensions(measurements);
817
- const baselineScores = measurements.map((measurement) => measurement.baseline.score);
818
- const candidateScores = measurements.map((measurement) => measurement.candidate.score);
819
- const {
820
- confidenceLevel: confidence,
821
- resamples,
822
- bootstrapSeed,
823
- deltaThreshold,
824
- minProductiveRuns,
825
- budgetUsd,
826
- criticalDimensions,
827
- regressionTolerance
828
- } = policy;
829
- const significance = heldoutSignificance(
830
- { before: baselineScores, after: candidateScores, cellIds: cellIds2 },
831
- {
832
- confidence,
833
- resamples,
834
- seed: bootstrapSeed,
835
- statistic: "mean",
836
- deltaThreshold,
837
- minProductiveRuns
838
- }
839
- );
840
- const overall = measuredEstimate(baselineScores, candidateScores, {
841
- confidence,
842
- resamples,
843
- seed: bootstrapSeed
844
- });
845
- const objectives = [
846
- {
847
- kind: "objective",
848
- name: "benchmark-score",
849
- direction: "higher-is-better",
850
- unit: "score",
851
- availability: "measured",
852
- ...overall
853
- },
854
- ...dimensions.map((name, index) => ({
855
- kind: "dimension",
856
- objective: "benchmark-score",
857
- name,
858
- direction: "higher-is-better",
859
- unit: "score",
860
- availability: "measured",
861
- ...measuredEstimate(
862
- measurements.map((measurement) => dimensionScore(measurement.baseline, name)),
863
- measurements.map((measurement) => dimensionScore(measurement.candidate, name)),
864
- {
865
- confidence,
866
- resamples,
867
- seed: bootstrapSeed + index + 1
868
- }
869
- )
870
- }))
871
- ];
872
- const cost = measuredEstimate(
873
- measurements.map((measurement) => measurement.baseline.costUsd),
874
- measurements.map((measurement) => measurement.candidate.costUsd),
875
- {
876
- confidence,
877
- resamples,
878
- seed: bootstrapSeed + dimensions.length + 1
879
- }
880
- );
881
- const latency = measuredEstimate(
882
- measurements.map((measurement) => measurement.baseline.latencyMs),
883
- measurements.map((measurement) => measurement.candidate.latencyMs),
884
- {
885
- confidence,
886
- resamples,
887
- seed: bootstrapSeed + dimensions.length + 2
888
- }
889
- );
890
- objectives.push(
891
- {
892
- kind: "cost",
893
- name: "cost",
894
- direction: "lower-is-better",
895
- unit: "usd",
896
- availability: "measured",
897
- ...cost
898
- },
899
- {
900
- kind: "latency",
901
- name: "latency",
902
- direction: "lower-is-better",
903
- unit: "milliseconds",
904
- availability: "measured",
905
- ...latency
906
- }
907
- );
908
- const power = baselineScores.length >= 3 ? powerPreflight({
909
- baselineComposites: baselineScores,
910
- pairedN: baselineScores.length,
911
- deltaThreshold,
912
- confidence,
913
- sharedScorerChannel: options.sharedScorerChannel
914
- }) : void 0;
915
- const powerSufficient = baselineScores.length >= minProductiveRuns && power !== void 0 && !power.underpowered;
916
- const guardedDimensions = new Set(criticalDimensions);
917
- const missingCriticalDimensions = criticalDimensions.filter(
918
- (dimension) => !dimensions.includes(dimension)
919
- );
920
- const regressions = objectives.filter(
921
- (objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance
922
- );
923
- const executionCostUsd = measurements.reduce(
924
- (sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd,
925
- 0
926
- );
927
- const executionDurationMs = measurements.reduce(
928
- (sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs,
929
- 0
930
- );
931
- const completedRuns = measurements.flatMap((measurement) => [
932
- measurement.baseline,
933
- measurement.candidate
934
- ]);
935
- const incompleteRuns = completedRuns.filter((run) => !run.completed);
936
- const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
937
- const totalCostUsd = executionCostUsd + additionalCostUsd;
938
- const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
939
- const checks = [
940
- { name: "paired-significance", passed: significance.significant },
941
- { name: "statistical-power", passed: powerSufficient },
942
- { name: "all-runs-completed", passed: incompleteRuns.length === 0 },
943
- { name: "candidate-task-pass", passed: failedCandidateResults.length === 0 },
944
- {
945
- name: "critical-dimensions",
946
- passed: regressions.length === 0 && missingCriticalDimensions.length === 0
947
- },
948
- { name: "budget", passed: budgetPassed }
949
- ];
950
- const shipped = checks.every((check) => check.passed);
951
- const reasons = [
952
- ...significance.significant ? [] : [
953
- significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`
954
- ],
955
- ...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
956
- ...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
957
- ...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
958
- ...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
959
- ...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
960
- ...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
961
- ];
962
- return {
963
- overall: {
964
- name: "composite",
965
- direction: "higher-is-better",
966
- unit: "score",
967
- ...overall
968
- },
969
- objectives,
970
- decision: {
971
- outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
972
- reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
973
- contributingChecks: checks
974
- },
975
- power: {
976
- sufficient: powerSufficient,
977
- n: baselineScores.length,
978
- minimumDetectableDelta: power?.mde ?? 1,
979
- confidenceLevel: confidence,
980
- scaleAssumed: power?.scaleAssumed ?? true,
981
- sharedScorerChannel: options.sharedScorerChannel,
982
- reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
983
- },
984
- executionCostUsd,
985
- totalCostUsd,
986
- executionDurationMs
987
- };
988
- }
650
+ if (options.measurements.length === 0) throw new Error("paired measurement evaluation requires at least one paired cell");
651
+ const additionalCostUsd = options.additionalCostUsd ?? 0;
652
+ if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
653
+ if (typeof options.sharedScorerChannel !== "boolean") throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
654
+ const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
655
+ const measurements = options.measurements.map((measurement, index) => projectPairedMeasurement(measurement, index, options.adapter));
656
+ const cellIds = measurements.map((measurement) => measurement.cellId);
657
+ if (new Set(cellIds).size !== cellIds.length) throw new Error("paired measurement evaluation cell ids must be unique");
658
+ const dimensions = sharedProjectedDimensions(measurements);
659
+ const baselineScores = measurements.map((measurement) => measurement.baseline.score);
660
+ const candidateScores = measurements.map((measurement) => measurement.candidate.score);
661
+ const { confidenceLevel: confidence, resamples, bootstrapSeed, deltaThreshold, minProductiveRuns, budgetUsd, criticalDimensions, regressionTolerance } = policy;
662
+ const significance = heldoutSignificance({
663
+ before: baselineScores,
664
+ after: candidateScores,
665
+ cellIds
666
+ }, {
667
+ confidence,
668
+ resamples,
669
+ seed: bootstrapSeed,
670
+ statistic: "mean",
671
+ deltaThreshold,
672
+ minProductiveRuns
673
+ });
674
+ const overall = measuredEstimate(baselineScores, candidateScores, {
675
+ confidence,
676
+ resamples,
677
+ seed: bootstrapSeed
678
+ });
679
+ const objectives = [{
680
+ kind: "objective",
681
+ name: "benchmark-score",
682
+ direction: "higher-is-better",
683
+ unit: "score",
684
+ availability: "measured",
685
+ ...overall
686
+ }, ...dimensions.map((name, index) => ({
687
+ kind: "dimension",
688
+ objective: "benchmark-score",
689
+ name,
690
+ direction: "higher-is-better",
691
+ unit: "score",
692
+ availability: "measured",
693
+ ...measuredEstimate(measurements.map((measurement) => dimensionScore(measurement.baseline, name)), measurements.map((measurement) => dimensionScore(measurement.candidate, name)), {
694
+ confidence,
695
+ resamples,
696
+ seed: bootstrapSeed + index + 1
697
+ })
698
+ }))];
699
+ const cost = measuredEstimate(measurements.map((measurement) => measurement.baseline.costUsd), measurements.map((measurement) => measurement.candidate.costUsd), {
700
+ confidence,
701
+ resamples,
702
+ seed: bootstrapSeed + dimensions.length + 1
703
+ });
704
+ const latency = measuredEstimate(measurements.map((measurement) => measurement.baseline.latencyMs), measurements.map((measurement) => measurement.candidate.latencyMs), {
705
+ confidence,
706
+ resamples,
707
+ seed: bootstrapSeed + dimensions.length + 2
708
+ });
709
+ objectives.push({
710
+ kind: "cost",
711
+ name: "cost",
712
+ direction: "lower-is-better",
713
+ unit: "usd",
714
+ availability: "measured",
715
+ ...cost
716
+ }, {
717
+ kind: "latency",
718
+ name: "latency",
719
+ direction: "lower-is-better",
720
+ unit: "milliseconds",
721
+ availability: "measured",
722
+ ...latency
723
+ });
724
+ const power = baselineScores.length >= 3 ? powerPreflight({
725
+ baselineComposites: baselineScores,
726
+ pairedN: baselineScores.length,
727
+ deltaThreshold,
728
+ confidence,
729
+ sharedScorerChannel: options.sharedScorerChannel
730
+ }) : void 0;
731
+ const powerSufficient = baselineScores.length >= minProductiveRuns && power !== void 0 && !power.underpowered;
732
+ const guardedDimensions = new Set(criticalDimensions);
733
+ const missingCriticalDimensions = criticalDimensions.filter((dimension) => !dimensions.includes(dimension));
734
+ const regressions = objectives.filter((objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance);
735
+ const executionCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
736
+ const executionDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
737
+ const incompleteRuns = measurements.flatMap((measurement) => [measurement.baseline, measurement.candidate]).filter((run) => !run.completed);
738
+ const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
739
+ const totalCostUsd = executionCostUsd + additionalCostUsd;
740
+ const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
741
+ const checks = [
742
+ {
743
+ name: "paired-significance",
744
+ passed: significance.significant
745
+ },
746
+ {
747
+ name: "statistical-power",
748
+ passed: powerSufficient
749
+ },
750
+ {
751
+ name: "all-runs-completed",
752
+ passed: incompleteRuns.length === 0
753
+ },
754
+ {
755
+ name: "candidate-task-pass",
756
+ passed: failedCandidateResults.length === 0
757
+ },
758
+ {
759
+ name: "critical-dimensions",
760
+ passed: regressions.length === 0 && missingCriticalDimensions.length === 0
761
+ },
762
+ {
763
+ name: "budget",
764
+ passed: budgetPassed
765
+ }
766
+ ];
767
+ const shipped = checks.every((check) => check.passed);
768
+ const reasons = [
769
+ ...significance.significant ? [] : [significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`],
770
+ ...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
771
+ ...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
772
+ ...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
773
+ ...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
774
+ ...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
775
+ ...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
776
+ ];
777
+ return {
778
+ overall: {
779
+ name: "composite",
780
+ direction: "higher-is-better",
781
+ unit: "score",
782
+ ...overall
783
+ },
784
+ objectives,
785
+ decision: {
786
+ outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
787
+ reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
788
+ contributingChecks: checks
789
+ },
790
+ power: {
791
+ sufficient: powerSufficient,
792
+ n: baselineScores.length,
793
+ minimumDetectableDelta: power?.mde ?? 1,
794
+ confidenceLevel: confidence,
795
+ scaleAssumed: power?.scaleAssumed ?? true,
796
+ sharedScorerChannel: options.sharedScorerChannel,
797
+ reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
798
+ },
799
+ executionCostUsd,
800
+ totalCostUsd,
801
+ executionDurationMs
802
+ };
803
+ }
804
+ /** Build the only publishable comparison: paired statistics over Runtime receipts. */
989
805
  function measuredComparisonFromCandidateExperiment(options) {
990
- const experiment = verifyCandidateExperiment(options.experiment);
991
- const measurements = options.measurements.map(
992
- (measurement, index) => verifyMeasurement(experiment, measurement, index)
993
- );
994
- const expectedN = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
995
- if (measurements.length !== expectedN) {
996
- throw new Error(
997
- `candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`
998
- );
999
- }
1000
- verifyStableProfileMaterialization(measurements);
1001
- if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
1002
- const searchCostUsd = options.searchCostUsd ?? 0;
1003
- const evaluation = evaluatePairedMeasurements({
1004
- measurements: measurements.map((measurement, index) => ({
1005
- cellId: cellIds(experiment)[index],
1006
- ...measurement
1007
- })),
1008
- policy: experiment.policy,
1009
- adapter: candidateExecutionEvidenceAdapter,
1010
- sharedScorerChannel: true,
1011
- additionalCostUsd: searchCostUsd
1012
- });
1013
- const diff = deriveCandidateBundleDiff(experiment);
1014
- const searchDurationMs = options.searchDurationMs ?? 0;
1015
- const totalCostUsd = evaluation.totalCostUsd;
1016
- const durationMs = evaluation.executionDurationMs + searchDurationMs;
1017
- const provisional = agentImprovementMeasuredComparisonSchema.parse({
1018
- kind: "agent-improvement-measured-comparison",
1019
- experiment,
1020
- measurements,
1021
- overall: evaluation.overall,
1022
- objectives: evaluation.objectives,
1023
- ...options.candidate ? { candidate: options.candidate } : {},
1024
- decision: evaluation.decision,
1025
- power: evaluation.power,
1026
- provenance: {
1027
- kind: "agent-eval-loop",
1028
- schema: "agent-candidate-experiment",
1029
- runId: options.runId,
1030
- recordDigest: canonicalCandidateDigest({}),
1031
- baselineContentHash: experiment.baseline.digest,
1032
- candidateContentHash: experiment.candidate.digest
1033
- },
1034
- diff,
1035
- evaluation: {
1036
- generationsExplored: options.generationsExplored ?? 0,
1037
- searchDurationMs,
1038
- executionDurationMs: evaluation.executionDurationMs,
1039
- durationMs,
1040
- searchCostUsd,
1041
- executionCostUsd: evaluation.executionCostUsd,
1042
- totalCostUsd
1043
- },
1044
- ...options.metadata ? { metadata: options.metadata } : {}
1045
- });
1046
- const { recordDigest: _recordDigest, ...provenance } = provisional.provenance;
1047
- return agentImprovementMeasuredComparisonSchema.parse({
1048
- ...provisional,
1049
- provenance: {
1050
- ...provenance,
1051
- recordDigest: canonicalCandidateDigest({ ...provisional, provenance })
1052
- }
1053
- });
1054
- }
806
+ const experiment = verifyCandidateExperiment(options.experiment);
807
+ const measurements = options.measurements.map((measurement, index) => verifyMeasurement(experiment, measurement, index));
808
+ const expectedN = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
809
+ if (measurements.length !== expectedN) throw new Error(`candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`);
810
+ verifyStableProfileMaterialization(measurements);
811
+ if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
812
+ const searchCostUsd = options.searchCostUsd ?? 0;
813
+ const evaluation = evaluatePairedMeasurements({
814
+ measurements: measurements.map((measurement, index) => ({
815
+ cellId: cellIds(experiment)[index],
816
+ ...measurement
817
+ })),
818
+ policy: experiment.policy,
819
+ adapter: candidateExecutionEvidenceAdapter,
820
+ sharedScorerChannel: true,
821
+ additionalCostUsd: searchCostUsd
822
+ });
823
+ const diff = deriveCandidateBundleDiff(experiment);
824
+ const searchDurationMs = options.searchDurationMs ?? 0;
825
+ const totalCostUsd = evaluation.totalCostUsd;
826
+ const durationMs = evaluation.executionDurationMs + searchDurationMs;
827
+ const provisional = agentImprovementMeasuredComparisonSchema.parse({
828
+ kind: "agent-improvement-measured-comparison",
829
+ experiment,
830
+ measurements,
831
+ overall: evaluation.overall,
832
+ objectives: evaluation.objectives,
833
+ ...options.candidate ? { candidate: options.candidate } : {},
834
+ decision: evaluation.decision,
835
+ power: evaluation.power,
836
+ provenance: {
837
+ kind: "agent-eval-loop",
838
+ schema: "agent-candidate-experiment",
839
+ runId: options.runId,
840
+ recordDigest: canonicalCandidateDigest({}),
841
+ baselineContentHash: experiment.baseline.digest,
842
+ candidateContentHash: experiment.candidate.digest
843
+ },
844
+ diff,
845
+ evaluation: {
846
+ generationsExplored: options.generationsExplored ?? 0,
847
+ searchDurationMs,
848
+ executionDurationMs: evaluation.executionDurationMs,
849
+ durationMs,
850
+ searchCostUsd,
851
+ executionCostUsd: evaluation.executionCostUsd,
852
+ totalCostUsd
853
+ },
854
+ ...options.metadata ? { metadata: options.metadata } : {}
855
+ });
856
+ const { recordDigest: _recordDigest, ...provenance } = provisional.provenance;
857
+ return agentImprovementMeasuredComparisonSchema.parse({
858
+ ...provisional,
859
+ provenance: {
860
+ ...provenance,
861
+ recordDigest: canonicalCandidateDigest({
862
+ ...provisional,
863
+ provenance
864
+ })
865
+ }
866
+ });
867
+ }
868
+ /** Recompute every statistic and decision from the signed experiment receipts. */
1055
869
  function verifyCandidateExperimentComparison(input) {
1056
- const comparison = agentImprovementMeasuredComparisonSchema.parse(input);
1057
- const recomputed = measuredComparisonFromCandidateExperiment({
1058
- experiment: comparison.experiment,
1059
- measurements: comparison.measurements,
1060
- runId: comparison.provenance.runId,
1061
- ...comparison.candidate ? { candidate: comparison.candidate } : {},
1062
- generationsExplored: comparison.evaluation.generationsExplored,
1063
- searchDurationMs: comparison.evaluation.searchDurationMs,
1064
- searchCostUsd: comparison.evaluation.searchCostUsd,
1065
- ...comparison.metadata ? { metadata: comparison.metadata } : {}
1066
- });
1067
- if (canonicalCandidateDigest(recomputed) !== canonicalCandidateDigest(comparison)) {
1068
- throw new Error("candidate experiment comparison does not match its Runtime receipts");
1069
- }
1070
- return comparison;
870
+ const comparison = agentImprovementMeasuredComparisonSchema.parse(input);
871
+ if (canonicalCandidateDigest(measuredComparisonFromCandidateExperiment({
872
+ experiment: comparison.experiment,
873
+ measurements: comparison.measurements,
874
+ runId: comparison.provenance.runId,
875
+ ...comparison.candidate ? { candidate: comparison.candidate } : {},
876
+ generationsExplored: comparison.evaluation.generationsExplored,
877
+ searchDurationMs: comparison.evaluation.searchDurationMs,
878
+ searchCostUsd: comparison.evaluation.searchCostUsd,
879
+ ...comparison.metadata ? { metadata: comparison.metadata } : {}
880
+ })) !== canonicalCandidateDigest(comparison)) throw new Error("candidate experiment comparison does not match its Runtime receipts");
881
+ return comparison;
1071
882
  }
1072
883
  function deriveCandidateBundleDiff(experiment) {
1073
- const surfaces = ["profile", "code", "execution", "knowledge", "memory"];
1074
- const changed = surfaces.flatMap((surface) => {
1075
- const baseline = experiment.baseline[surface] ?? null;
1076
- const candidate = experiment.candidate[surface] ?? null;
1077
- const baselineDigest = canonicalCandidateDigest(baseline);
1078
- const candidateDigest = canonicalCandidateDigest(candidate);
1079
- if (baselineDigest === candidateDigest) return [];
1080
- return [
1081
- [
1082
- `--- baseline/${surface} (${baselineDigest})`,
1083
- `+++ candidate/${surface} (${candidateDigest})`,
1084
- JSON.stringify({ baseline, candidate }, null, 2)
1085
- ].join("\n")
1086
- ];
1087
- });
1088
- if (changed.length === 0) {
1089
- throw new Error("candidate experiment has no changed candidate surface");
1090
- }
1091
- return changed.join("\n\n");
884
+ const changed = [
885
+ "profile",
886
+ "code",
887
+ "execution",
888
+ "knowledge",
889
+ "memory"
890
+ ].flatMap((surface) => {
891
+ const baseline = experiment.baseline[surface] ?? null;
892
+ const candidate = experiment.candidate[surface] ?? null;
893
+ const baselineDigest = canonicalCandidateDigest(baseline);
894
+ const candidateDigest = canonicalCandidateDigest(candidate);
895
+ if (baselineDigest === candidateDigest) return [];
896
+ return [[
897
+ `--- baseline/${surface} (${baselineDigest})`,
898
+ `+++ candidate/${surface} (${candidateDigest})`,
899
+ JSON.stringify({
900
+ baseline,
901
+ candidate
902
+ }, null, 2)
903
+ ].join("\n")];
904
+ });
905
+ if (changed.length === 0) throw new Error("candidate experiment has no changed candidate surface");
906
+ return changed.join("\n\n");
1092
907
  }
1093
908
  function verifyCandidateBenchmarkTask(input) {
1094
- const task = agentCandidateBenchmarkTaskSchema.parse(input);
1095
- verifySelfAddressed(task, "candidate benchmark task");
1096
- return task;
909
+ const task = agentCandidateBenchmarkTaskSchema.parse(input);
910
+ verifySelfAddressed(task, "candidate benchmark task");
911
+ return task;
1097
912
  }
1098
913
  function verifyCandidateBenchmarkSuiteInputs(input) {
1099
- if (input === null || typeof input !== "object" || Array.isArray(input)) {
1100
- throw new Error("candidate benchmark suite inputs must be an object");
1101
- }
1102
- const candidate = input;
1103
- const suite = verifyCandidateBenchmarkSuite(candidate.suite);
1104
- if (!Array.isArray(candidate.tasks) || candidate.tasks.length !== suite.taskDigests.length) {
1105
- throw new Error("candidate benchmark suite task count does not match its signed digests");
1106
- }
1107
- candidate.tasks.forEach((task, index) => {
1108
- const verified = verifyCandidateBenchmarkTask(task);
1109
- if (verified.digest !== suite.taskDigests[index]) {
1110
- throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
1111
- }
1112
- });
1113
- return { suite, tasks: candidate.tasks };
914
+ if (input === null || typeof input !== "object" || Array.isArray(input)) throw new Error("candidate benchmark suite inputs must be an object");
915
+ const candidate = input;
916
+ const suite = verifyCandidateBenchmarkSuite(candidate.suite);
917
+ if (!Array.isArray(candidate.tasks) || candidate.tasks.length !== suite.taskDigests.length) throw new Error("candidate benchmark suite task count does not match its signed digests");
918
+ candidate.tasks.forEach((task, index) => {
919
+ if (verifyCandidateBenchmarkTask(task).digest !== suite.taskDigests[index]) throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
920
+ });
921
+ return {
922
+ suite,
923
+ tasks: candidate.tasks
924
+ };
1114
925
  }
1115
926
  function verifyCandidateBenchmarkSuite(input) {
1116
- const suite = agentCandidateBenchmarkSuiteSchema.parse(input);
1117
- verifySelfAddressed(suite, "candidate benchmark suite");
1118
- return suite;
927
+ const suite = agentCandidateBenchmarkSuiteSchema.parse(input);
928
+ verifySelfAddressed(suite, "candidate benchmark suite");
929
+ return suite;
1119
930
  }
1120
931
  function verifyBundle(input, label) {
1121
- const bundle = agentCandidateBundleSchema.parse(input);
1122
- verifySelfAddressed(bundle, label);
1123
- return bundle;
932
+ const bundle = agentCandidateBundleSchema.parse(input);
933
+ verifySelfAddressed(bundle, label);
934
+ return bundle;
1124
935
  }
1125
936
  function verifyMeasurement(experiment, input, index) {
1126
- const suite = experiment.benchmark.suite;
1127
- const taskIndex = Math.floor(index / suite.reps);
1128
- const repetition = index % suite.reps;
1129
- const task = experiment.benchmark.tasks[taskIndex];
1130
- const seed = suite.seeds[index];
1131
- if (!task || seed === void 0) {
1132
- throw new Error(`candidate experiment measurement ${index} is outside the signed suite`);
1133
- }
1134
- const baseline = verifyExecutionEvidence(input.baseline);
1135
- const candidate = verifyExecutionEvidence(input.candidate);
1136
- for (const [arm, evidence] of [
1137
- ["baseline", baseline],
1138
- ["candidate", candidate]
1139
- ]) {
1140
- const bundle = experiment[arm];
1141
- const materialization = evidence.materializationReceipt;
1142
- const plan = materialization.executionPlan.material;
1143
- const runCell = plan.runCell;
1144
- verifySelfAddressed(runCell, "candidate run cell");
1145
- if (runCell.experimentDigest !== experiment.digest || runCell.arm !== arm || runCell.bundleDigest !== bundle.digest || runCell.suiteDigest !== suite.digest || runCell.taskDigest !== task.digest || runCell.taskIndex !== taskIndex || runCell.repetition !== repetition || runCell.seed !== seed || runCell.attempt > task.attempt.maxAttempts || materialization.bundleDigest !== bundle.digest || materialization.benchmark.suite.digest !== suite.digest || materialization.benchmark.task.digest !== task.digest || materialization.codeKind !== bundle.code.kind || materialization.profileActivation.profilePlan.material.sourceProfileDigest !== canonicalCandidateDigest(bundle.profile) || evidence.receipt.runCellDigest !== runCell.digest || JSON.stringify(materialization.resolvedModel) !== JSON.stringify(task.model)) {
1146
- throw new Error(`candidate experiment measurement ${index} substituted its ${arm} arm`);
1147
- }
1148
- verifyTaskOutcome(task, evidence, index, arm);
1149
- }
1150
- const baselinePlan = baseline.materializationReceipt.executionPlan.material;
1151
- const candidatePlan = candidate.materializationReceipt.executionPlan.material;
1152
- if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) {
1153
- throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
1154
- }
1155
- return { baseline, candidate };
937
+ const suite = experiment.benchmark.suite;
938
+ const taskIndex = Math.floor(index / suite.reps);
939
+ const repetition = index % suite.reps;
940
+ const task = experiment.benchmark.tasks[taskIndex];
941
+ const seed = suite.seeds[index];
942
+ if (!task || seed === void 0) throw new Error(`candidate experiment measurement ${index} is outside the signed suite`);
943
+ const baseline = verifyExecutionEvidence(input.baseline);
944
+ const candidate = verifyExecutionEvidence(input.candidate);
945
+ for (const [arm, evidence] of [["baseline", baseline], ["candidate", candidate]]) {
946
+ const bundle = experiment[arm];
947
+ const materialization = evidence.materializationReceipt;
948
+ const runCell = materialization.executionPlan.material.runCell;
949
+ verifySelfAddressed(runCell, "candidate run cell");
950
+ if (runCell.experimentDigest !== experiment.digest || runCell.arm !== arm || runCell.bundleDigest !== bundle.digest || runCell.suiteDigest !== suite.digest || runCell.taskDigest !== task.digest || runCell.taskIndex !== taskIndex || runCell.repetition !== repetition || runCell.seed !== seed || runCell.attempt > task.attempt.maxAttempts || materialization.bundleDigest !== bundle.digest || materialization.benchmark.suite.digest !== suite.digest || materialization.benchmark.task.digest !== task.digest || materialization.codeKind !== bundle.code.kind || materialization.profileActivation.profilePlan.material.sourceProfileDigest !== canonicalCandidateDigest(bundle.profile) || evidence.receipt.runCellDigest !== runCell.digest || JSON.stringify(materialization.resolvedModel) !== JSON.stringify(task.model)) throw new Error(`candidate experiment measurement ${index} substituted its ${arm} arm`);
951
+ verifyTaskOutcome(task, evidence, index, arm);
952
+ }
953
+ const baselinePlan = baseline.materializationReceipt.executionPlan.material;
954
+ const candidatePlan = candidate.materializationReceipt.executionPlan.material;
955
+ if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
956
+ return {
957
+ baseline,
958
+ candidate
959
+ };
1156
960
  }
1157
961
  function verifyExecutionEvidence(input) {
1158
- const evidence = candidateExecutionEvidenceSchema.parse(input);
1159
- verifySelfAddressed(evidence, "candidate execution evidence");
1160
- verifySelfAddressed(evidence.materializationReceipt, "candidate materialization receipt");
1161
- verifySelfAddressed(
1162
- evidence.materializationReceipt.profileActivation,
1163
- "candidate profile activation"
1164
- );
1165
- verifyMaterialAddressed(
1166
- evidence.materializationReceipt.profileActivation.profilePlan,
1167
- "candidate profile plan"
1168
- );
1169
- verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
1170
- verifySelfAddressed(evidence.receipt, "candidate run receipt");
1171
- verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
1172
- verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
1173
- verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
1174
- return evidence;
962
+ const evidence = candidateExecutionEvidenceSchema.parse(input);
963
+ verifySelfAddressed(evidence, "candidate execution evidence");
964
+ verifySelfAddressed(evidence.materializationReceipt, "candidate materialization receipt");
965
+ verifySelfAddressed(evidence.materializationReceipt.profileActivation, "candidate profile activation");
966
+ verifyMaterialAddressed(evidence.materializationReceipt.profileActivation.profilePlan, "candidate profile plan");
967
+ verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
968
+ verifySelfAddressed(evidence.receipt, "candidate run receipt");
969
+ verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
970
+ verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
971
+ verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
972
+ return evidence;
1175
973
  }
1176
974
  function verifySelfAddressed(document, label) {
1177
- if (canonicalCandidateDigest(omitTopLevelDigest(document)) !== document.digest) {
1178
- throw new Error(`${label} digest is invalid`);
1179
- }
975
+ if (canonicalCandidateDigest(omitTopLevelDigest(document)) !== document.digest) throw new Error(`${label} digest is invalid`);
1180
976
  }
1181
977
  function verifyTaskOutcome(task, evidence, index, arm) {
1182
- const outcome = evidence.receipt.taskOutcome.material.outcome;
1183
- const result = evidence.receipt.benchmarkResult.material;
1184
- const prefix = `candidate experiment measurement ${index} ${arm}`;
1185
- if (result.evidence.sha256 === task.grader.artifact.sha256) {
1186
- throw new Error(`${prefix} reused grader bytes as grading evidence`);
1187
- }
1188
- const usage = combinedUsage(evidence);
1189
- const usageChecks = [
1190
- [usage.modelCalls, task.limits.maxModelCalls, "model calls"],
1191
- [usage.inputTokens, task.limits.maxInputTokens, "input tokens"],
1192
- [usage.outputTokens, task.limits.maxOutputTokens, "output tokens"],
1193
- [usage.costUsdNanos, Math.round(task.limits.maxCostUsd * 1e9), "cost"]
1194
- ];
1195
- for (const [actual, maximum, label] of usageChecks) {
1196
- if (actual > maximum) {
1197
- throw new Error(`${prefix} ${label} ${actual} exceeds the signed limit ${maximum}`);
1198
- }
1199
- }
1200
- if (outcome.kind !== task.outcome.kind) {
1201
- throw new Error(`${prefix} returned an outcome outside the signed task contract`);
1202
- }
1203
- if (task.outcome.kind === "output") {
1204
- if (outcome.kind !== "output" || outcome.spec.mediaType !== task.outcome.mediaType || outcome.spec.maxBytes !== task.outcome.maxBytes) {
1205
- throw new Error(`${prefix} changed the signed output contract`);
1206
- }
1207
- return;
1208
- }
1209
- const repository = task.repository;
1210
- if (outcome.kind !== "workspace" || repository === void 0 || outcome.baseRepository.identity !== repository.identity || outcome.baseRepository.rootIdentity !== repository.rootIdentity || outcome.baseRepository.commit !== repository.baseCommit || outcome.baseRepository.tree !== repository.baseTree) {
1211
- throw new Error(`${prefix} did not start from the signed repository state`);
1212
- }
978
+ const outcome = evidence.receipt.taskOutcome.material.outcome;
979
+ const result = evidence.receipt.benchmarkResult.material;
980
+ const prefix = `candidate experiment measurement ${index} ${arm}`;
981
+ if (result.evidence.sha256 === task.grader.artifact.sha256) throw new Error(`${prefix} reused grader bytes as grading evidence`);
982
+ const usage = combinedUsage(evidence);
983
+ const usageChecks = [
984
+ [
985
+ usage.modelCalls,
986
+ task.limits.maxModelCalls,
987
+ "model calls"
988
+ ],
989
+ [
990
+ usage.inputTokens,
991
+ task.limits.maxInputTokens,
992
+ "input tokens"
993
+ ],
994
+ [
995
+ usage.outputTokens,
996
+ task.limits.maxOutputTokens,
997
+ "output tokens"
998
+ ],
999
+ [
1000
+ usage.costUsdNanos,
1001
+ Math.round(task.limits.maxCostUsd * 1e9),
1002
+ "cost"
1003
+ ]
1004
+ ];
1005
+ for (const [actual, maximum, label] of usageChecks) if (actual > maximum) throw new Error(`${prefix} ${label} ${actual} exceeds the signed limit ${maximum}`);
1006
+ if (outcome.kind !== task.outcome.kind) throw new Error(`${prefix} returned an outcome outside the signed task contract`);
1007
+ if (task.outcome.kind === "output") {
1008
+ if (outcome.kind !== "output" || outcome.spec.mediaType !== task.outcome.mediaType || outcome.spec.maxBytes !== task.outcome.maxBytes) throw new Error(`${prefix} changed the signed output contract`);
1009
+ return;
1010
+ }
1011
+ const repository = task.repository;
1012
+ if (outcome.kind !== "workspace" || repository === void 0 || outcome.baseRepository.identity !== repository.identity || outcome.baseRepository.rootIdentity !== repository.rootIdentity || outcome.baseRepository.commit !== repository.baseCommit || outcome.baseRepository.tree !== repository.baseTree) throw new Error(`${prefix} did not start from the signed repository state`);
1213
1013
  }
1214
1014
  function verifyStableProfileMaterialization(measurements) {
1215
- for (const arm of ["baseline", "candidate"]) {
1216
- const expected = measurements[0]?.[arm].materializationReceipt.profileActivation;
1217
- if (!expected) throw new Error("candidate experiment contains no profile materialization");
1218
- const expectedDigest = canonicalCandidateDigest({
1219
- profilePlanDigest: expected.profilePlan.digest,
1220
- files: expected.files
1221
- });
1222
- for (const [index, measurement] of measurements.entries()) {
1223
- const activation = measurement[arm].materializationReceipt.profileActivation;
1224
- if (canonicalCandidateDigest({
1225
- profilePlanDigest: activation.profilePlan.digest,
1226
- files: activation.files
1227
- }) !== expectedDigest) {
1228
- throw new Error(
1229
- `candidate experiment measurement ${index} ${arm} materialized a different profile`
1230
- );
1231
- }
1232
- }
1233
- }
1015
+ for (const arm of ["baseline", "candidate"]) {
1016
+ const expected = measurements[0]?.[arm].materializationReceipt.profileActivation;
1017
+ if (!expected) throw new Error("candidate experiment contains no profile materialization");
1018
+ const expectedDigest = canonicalCandidateDigest({
1019
+ profilePlanDigest: expected.profilePlan.digest,
1020
+ files: expected.files
1021
+ });
1022
+ for (const [index, measurement] of measurements.entries()) {
1023
+ const activation = measurement[arm].materializationReceipt.profileActivation;
1024
+ if (canonicalCandidateDigest({
1025
+ profilePlanDigest: activation.profilePlan.digest,
1026
+ files: activation.files
1027
+ }) !== expectedDigest) throw new Error(`candidate experiment measurement ${index} ${arm} materialized a different profile`);
1028
+ }
1029
+ }
1234
1030
  }
1235
1031
  function completedSuccessfully(evidence) {
1236
- const termination = evidence.receipt.termination;
1237
- return termination.kind === "exit" && termination.exitCode === 0;
1032
+ const termination = evidence.receipt.termination;
1033
+ return termination.kind === "exit" && termination.exitCode === 0;
1238
1034
  }
1239
1035
  function verifyMaterialAddressed(evidence, label) {
1240
- if (canonicalCandidateDigest(evidence.material) !== evidence.digest) {
1241
- throw new Error(`${label} digest is invalid`);
1242
- }
1036
+ if (canonicalCandidateDigest(evidence.material) !== evidence.digest) throw new Error(`${label} digest is invalid`);
1243
1037
  }
1244
1038
  function projectPairedMeasurement(measurement, index, adapter) {
1245
- if (typeof measurement.cellId !== "string" || !measurement.cellId.trim()) {
1246
- throw new Error(`paired measurement ${index} requires a cell id`);
1247
- }
1248
- return {
1249
- cellId: measurement.cellId,
1250
- baseline: projectRun(measurement.baseline, adapter, `paired measurement ${index} baseline`),
1251
- candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
1252
- };
1039
+ if (typeof measurement.cellId !== "string" || !measurement.cellId.trim()) throw new Error(`paired measurement ${index} requires a cell id`);
1040
+ return {
1041
+ cellId: measurement.cellId,
1042
+ baseline: projectRun(measurement.baseline, adapter, `paired measurement ${index} baseline`),
1043
+ candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
1044
+ };
1253
1045
  }
1254
1046
  function projectRun(run, adapter, label) {
1255
- const suppliedDimensions = adapter.dimensions(run);
1256
- if (!Array.isArray(suppliedDimensions)) {
1257
- throw new Error(`${label} dimensions must be an array`);
1258
- }
1259
- const dimensions = /* @__PURE__ */ new Map();
1260
- for (const dimension of suppliedDimensions) {
1261
- if (typeof dimension.name !== "string" || !dimension.name.trim()) {
1262
- throw new Error(`${label} contains an unnamed dimension`);
1263
- }
1264
- if (dimensions.has(dimension.name)) {
1265
- throw new Error(`${label} repeats dimension '${dimension.name}'`);
1266
- }
1267
- dimensions.set(dimension.name, finiteMeasurement(dimension.score, `${label} ${dimension.name}`));
1268
- }
1269
- const completed = adapter.completed(run);
1270
- const passed = adapter.passed(run);
1271
- if (typeof completed !== "boolean" || typeof passed !== "boolean") {
1272
- throw new Error(`${label} completion and pass values must be booleans`);
1273
- }
1274
- return {
1275
- score: finiteMeasurement(adapter.score(run), `${label} score`),
1276
- dimensions,
1277
- costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
1278
- latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
1279
- completed,
1280
- passed
1281
- };
1047
+ const suppliedDimensions = adapter.dimensions(run);
1048
+ if (!Array.isArray(suppliedDimensions)) throw new Error(`${label} dimensions must be an array`);
1049
+ const dimensions = /* @__PURE__ */ new Map();
1050
+ for (const dimension of suppliedDimensions) {
1051
+ if (typeof dimension.name !== "string" || !dimension.name.trim()) throw new Error(`${label} contains an unnamed dimension`);
1052
+ if (dimensions.has(dimension.name)) throw new Error(`${label} repeats dimension '${dimension.name}'`);
1053
+ dimensions.set(dimension.name, finiteMeasurement(dimension.score, `${label} ${dimension.name}`));
1054
+ }
1055
+ const completed = adapter.completed(run);
1056
+ const passed = adapter.passed(run);
1057
+ if (typeof completed !== "boolean" || typeof passed !== "boolean") throw new Error(`${label} completion and pass values must be booleans`);
1058
+ return {
1059
+ score: finiteMeasurement(adapter.score(run), `${label} score`),
1060
+ dimensions,
1061
+ costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
1062
+ latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
1063
+ completed,
1064
+ passed
1065
+ };
1282
1066
  }
1283
1067
  function sharedProjectedDimensions(measurements) {
1284
- const expected = [...measurements[0].baseline.dimensions.keys()];
1285
- for (const [index, measurement] of measurements.entries()) {
1286
- for (const [arm, run] of [
1287
- ["baseline", measurement.baseline],
1288
- ["candidate", measurement.candidate]
1289
- ]) {
1290
- const actual = [...run.dimensions.keys()];
1291
- if (JSON.stringify(actual) !== JSON.stringify(expected)) {
1292
- throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
1293
- }
1294
- }
1295
- }
1296
- return expected;
1068
+ const expected = [...measurements[0].baseline.dimensions.keys()];
1069
+ for (const [index, measurement] of measurements.entries()) for (const [arm, run] of [["baseline", measurement.baseline], ["candidate", measurement.candidate]]) {
1070
+ const actual = [...run.dimensions.keys()];
1071
+ if (JSON.stringify(actual) !== JSON.stringify(expected)) throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
1072
+ }
1073
+ return expected;
1297
1074
  }
1298
1075
  function dimensionScore(run, name) {
1299
- const value = run.dimensions.get(name);
1300
- if (value === void 0) throw new Error(`paired measurement is missing dimension '${name}'`);
1301
- return value;
1076
+ const value = run.dimensions.get(name);
1077
+ if (value === void 0) throw new Error(`paired measurement is missing dimension '${name}'`);
1078
+ return value;
1302
1079
  }
1303
1080
  function measuredEstimate(baseline, candidate, options) {
1304
- const bootstrap = pairedBootstrap(baseline, candidate, {
1305
- confidence: options.confidence,
1306
- resamples: options.resamples,
1307
- statistic: "mean",
1308
- seed: options.seed
1309
- });
1310
- const baselineMean = mean(baseline);
1311
- const candidateMean = mean(candidate);
1312
- const delta = candidateMean - baselineMean;
1313
- return {
1314
- baseline: baselineMean,
1315
- candidate: candidateMean,
1316
- delta,
1317
- confidenceInterval: {
1318
- level: bootstrap.confidence,
1319
- lower: Math.min(bootstrap.low, delta),
1320
- upper: Math.max(bootstrap.high, delta),
1321
- method: "paired-bootstrap",
1322
- statistic: "mean",
1323
- resamples: bootstrap.resamples
1324
- },
1325
- n: bootstrap.n
1326
- };
1081
+ const bootstrap = pairedBootstrap(baseline, candidate, {
1082
+ confidence: options.confidence,
1083
+ resamples: options.resamples,
1084
+ statistic: "mean",
1085
+ seed: options.seed
1086
+ });
1087
+ const baselineMean = mean(baseline);
1088
+ const candidateMean = mean(candidate);
1089
+ const delta = candidateMean - baselineMean;
1090
+ return {
1091
+ baseline: baselineMean,
1092
+ candidate: candidateMean,
1093
+ delta,
1094
+ confidenceInterval: {
1095
+ level: bootstrap.confidence,
1096
+ lower: Math.min(bootstrap.low, delta),
1097
+ upper: Math.max(bootstrap.high, delta),
1098
+ method: "paired-bootstrap",
1099
+ statistic: "mean",
1100
+ resamples: bootstrap.resamples
1101
+ },
1102
+ n: bootstrap.n
1103
+ };
1327
1104
  }
1328
1105
  function finiteMeasurement(value, label) {
1329
- if (!Number.isFinite(value)) throw new Error(`${label} must be finite`);
1330
- return value;
1106
+ if (!Number.isFinite(value)) throw new Error(`${label} must be finite`);
1107
+ return value;
1331
1108
  }
1332
1109
  function nonNegativeMeasurement(value, label) {
1333
- if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be non-negative`);
1334
- return value;
1335
- }
1336
- var candidateExecutionEvidenceAdapter = {
1337
- score: (evidence) => evidence.receipt.benchmarkResult.material.score,
1338
- dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
1339
- costUsd: costFromEvidence,
1340
- latencyMs: latencyFromEvidence,
1341
- completed: completedSuccessfully,
1342
- passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
1110
+ if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be non-negative`);
1111
+ return value;
1112
+ }
1113
+ const candidateExecutionEvidenceAdapter = {
1114
+ score: (evidence) => evidence.receipt.benchmarkResult.material.score,
1115
+ dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
1116
+ costUsd: costFromEvidence,
1117
+ latencyMs: latencyFromEvidence,
1118
+ completed: completedSuccessfully,
1119
+ passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
1343
1120
  };
1344
1121
  function costFromEvidence(evidence) {
1345
- return combinedUsage(evidence).costUsdNanos / 1e9;
1122
+ return combinedUsage(evidence).costUsdNanos / 1e9;
1346
1123
  }
1347
1124
  function latencyFromEvidence(evidence) {
1348
- return evidence.receipt.timing.durationMs + evidence.receipt.benchmarkResult.material.grading.timing.durationMs;
1125
+ return evidence.receipt.timing.durationMs + evidence.receipt.benchmarkResult.material.grading.timing.durationMs;
1349
1126
  }
1350
1127
  function combinedUsage(evidence) {
1351
- const candidate = evidence.receipt.modelSettlement.material.usage;
1352
- const grader = evidence.receipt.benchmarkResult.material.grading.usage;
1353
- return {
1354
- inputTokens: candidate.inputTokens + grader.inputTokens,
1355
- outputTokens: candidate.outputTokens + grader.outputTokens,
1356
- cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
1357
- reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
1358
- modelCalls: candidate.modelCalls + grader.modelCalls,
1359
- costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
1360
- };
1128
+ const candidate = evidence.receipt.modelSettlement.material.usage;
1129
+ const grader = evidence.receipt.benchmarkResult.material.grading.usage;
1130
+ return {
1131
+ inputTokens: candidate.inputTokens + grader.inputTokens,
1132
+ outputTokens: candidate.outputTokens + grader.outputTokens,
1133
+ cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
1134
+ reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
1135
+ modelCalls: candidate.modelCalls + grader.modelCalls,
1136
+ costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
1137
+ };
1361
1138
  }
1362
1139
  function cellIds(experiment) {
1363
- const { suite, tasks } = experiment.benchmark;
1364
- return suite.seeds.map((_, index) => {
1365
- const taskIndex = Math.floor(index / suite.reps);
1366
- const repetition = index % suite.reps;
1367
- return `${tasks[taskIndex]?.scenario.id ?? taskIndex}:${repetition}`;
1368
- });
1140
+ const { suite, tasks } = experiment.benchmark;
1141
+ return suite.seeds.map((_, index) => {
1142
+ const taskIndex = Math.floor(index / suite.reps);
1143
+ const repetition = index % suite.reps;
1144
+ return `${tasks[taskIndex]?.scenario.id ?? taskIndex}:${repetition}`;
1145
+ });
1369
1146
  }
1370
1147
  function mean(values) {
1371
- if (values.length === 0) throw new Error("candidate experiment requires measured values");
1372
- return values.reduce((sum, value) => sum + value, 0) / values.length;
1148
+ if (values.length === 0) throw new Error("candidate experiment requires measured values");
1149
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
1373
1150
  }
1374
1151
  function abortError(signal) {
1375
- return signal.reason instanceof Error ? signal.reason : new Error("candidate experiment aborted");
1376
- }
1377
-
1378
- // src/contract/eval-reporting-suite.ts
1379
- import { mkdir, writeFile } from "fs/promises";
1380
- import { dirname, join as join2 } from "path";
1381
-
1382
- // src/contract/intake/run-record-dir.ts
1383
- import { readdir, readFile, stat } from "fs/promises";
1384
- import { join } from "path";
1385
- var ANALYSIS_ARTIFACT = "analysis.json";
1152
+ return signal.reason instanceof Error ? signal.reason : /* @__PURE__ */ new Error("candidate experiment aborted");
1153
+ }
1154
+ //#endregion
1155
+ //#region src/contract/intake/run-record-dir.ts
1156
+ /**
1157
+ * # `intake/run-record-dir` load a directory or file of `RunRecord`s.
1158
+ *
1159
+ * The on-disk counterpart to the in-memory intake adapters: point it at a
1160
+ * single `.json` (array) / `.jsonl` (one record per line) file or at a
1161
+ * directory of such files, and it returns the substrate-canonical
1162
+ * `RunRecord[]` ready for `analyzeRuns({ runs })`.
1163
+ *
1164
+ * Validation is at the boundary: each parsed object goes through
1165
+ * `parseRunRecordSafe`. By default an invalid record fails loud with its
1166
+ * file + index; pass `onInvalid: 'collect'` to keep the valid records and
1167
+ * receive the rejects as structured diagnostics instead.
1168
+ */
1169
+ const ANALYSIS_ARTIFACT$1 = "analysis.json";
1386
1170
  function defaultInclude(fileName) {
1387
- if (fileName === ANALYSIS_ARTIFACT) return false;
1388
- return fileName.endsWith(".json") || fileName.endsWith(".jsonl");
1389
- }
1171
+ if (fileName === ANALYSIS_ARTIFACT$1) return false;
1172
+ return fileName.endsWith(".json") || fileName.endsWith(".jsonl");
1173
+ }
1174
+ /**
1175
+ * Resolve a file or directory path into validated `RunRecord[]`.
1176
+ *
1177
+ * A `.json` file must parse to a top-level array; a `.jsonl` file is one
1178
+ * record per non-empty line. Directories are read shallowly by default
1179
+ * (set `recursive` to descend); the `analysis.json` output artifact is
1180
+ * always excluded.
1181
+ */
1390
1182
  async function fromRunRecordDir(path, options = {}) {
1391
- const onInvalid = options.onInvalid ?? "throw";
1392
- const include = options.include ?? defaultInclude;
1393
- const stats = await stat(path);
1394
- const filePaths = stats.isDirectory() ? await collectFiles(path, include, options.recursive ?? false) : [path];
1395
- const runs = [];
1396
- const rejected = [];
1397
- for (const file of filePaths) {
1398
- const raw = await parseRecordFile(file);
1399
- for (const { index, value } of raw) {
1400
- const parsed = parseRunRecordSafe(value);
1401
- if (parsed.ok) {
1402
- runs.push(parsed.value);
1403
- continue;
1404
- }
1405
- const rejection = { file, index, reason: parsed.error.message };
1406
- if (onInvalid === "throw") {
1407
- throw new Error(
1408
- `fromRunRecordDir: invalid RunRecord in '${file}' at index ${index}: ${parsed.error.message}`
1409
- );
1410
- }
1411
- rejected.push(rejection);
1412
- }
1413
- }
1414
- return { runs, rejected, files: filePaths };
1415
- }
1183
+ const onInvalid = options.onInvalid ?? "throw";
1184
+ const include = options.include ?? defaultInclude;
1185
+ const filePaths = (await stat(path)).isDirectory() ? await collectFiles(path, include, options.recursive ?? false) : [path];
1186
+ const runs = [];
1187
+ const rejected = [];
1188
+ for (const file of filePaths) {
1189
+ const raw = await parseRecordFile(file);
1190
+ for (const { index, value } of raw) {
1191
+ const parsed = parseRunRecordSafe(value);
1192
+ if (parsed.ok) {
1193
+ runs.push(parsed.value);
1194
+ continue;
1195
+ }
1196
+ const rejection = {
1197
+ file,
1198
+ index,
1199
+ reason: parsed.error.message
1200
+ };
1201
+ if (onInvalid === "throw") throw new Error(`fromRunRecordDir: invalid RunRecord in '${file}' at index ${index}: ${parsed.error.message}`);
1202
+ rejected.push(rejection);
1203
+ }
1204
+ }
1205
+ return {
1206
+ runs,
1207
+ rejected,
1208
+ files: filePaths
1209
+ };
1210
+ }
1211
+ /** Read a single `.json` / `.jsonl` file into `{ index, value }` pairs. A
1212
+ * malformed JSONL line throws with its line number rather than being skipped —
1213
+ * silent line-dropping is how corpora quietly shrink. */
1416
1214
  async function parseRecordFile(file) {
1417
- const text = await readFile(file, "utf8");
1418
- const trimmed = text.trim();
1419
- if (trimmed.length === 0) return [];
1420
- if (trimmed.startsWith("[")) {
1421
- const parsed = JSON.parse(trimmed);
1422
- if (!Array.isArray(parsed)) {
1423
- throw new Error(`fromRunRecordDir: file '${file}' did not parse to an array`);
1424
- }
1425
- return parsed.map((value, index) => ({ index, value }));
1426
- }
1427
- const out = [];
1428
- const lines = trimmed.split("\n");
1429
- for (let i = 0; i < lines.length; i++) {
1430
- const line = lines[i].trim();
1431
- if (line.length === 0) continue;
1432
- try {
1433
- out.push({ index: i, value: JSON.parse(line) });
1434
- } catch (err) {
1435
- throw new Error(
1436
- `fromRunRecordDir: file '${file}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`
1437
- );
1438
- }
1439
- }
1440
- return out;
1441
- }
1215
+ const trimmed = (await readFile(file, "utf8")).trim();
1216
+ if (trimmed.length === 0) return [];
1217
+ if (trimmed.startsWith("[")) {
1218
+ const parsed = JSON.parse(trimmed);
1219
+ if (!Array.isArray(parsed)) throw new Error(`fromRunRecordDir: file '${file}' did not parse to an array`);
1220
+ return parsed.map((value, index) => ({
1221
+ index,
1222
+ value
1223
+ }));
1224
+ }
1225
+ const out = [];
1226
+ const lines = trimmed.split("\n");
1227
+ for (let i = 0; i < lines.length; i++) {
1228
+ const line = lines[i].trim();
1229
+ if (line.length === 0) continue;
1230
+ try {
1231
+ out.push({
1232
+ index: i,
1233
+ value: JSON.parse(line)
1234
+ });
1235
+ } catch (err) {
1236
+ throw new Error(`fromRunRecordDir: file '${file}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`);
1237
+ }
1238
+ }
1239
+ return out;
1240
+ }
1241
+ /** Sorted file list under a directory, filtered by `include`. Sorted so the
1242
+ * resulting `RunRecord` order — and any downstream fingerprint — is stable
1243
+ * across filesystems. */
1442
1244
  async function collectFiles(dir, include, recursive) {
1443
- const entries = await readdir(dir, { withFileTypes: true });
1444
- const files = [];
1445
- const subdirs = [];
1446
- for (const entry of entries) {
1447
- if (entry.isDirectory()) {
1448
- if (recursive) subdirs.push(join(dir, entry.name));
1449
- continue;
1450
- }
1451
- if (include(entry.name)) files.push(join(dir, entry.name));
1452
- }
1453
- files.sort();
1454
- subdirs.sort();
1455
- for (const sub of subdirs) {
1456
- files.push(...await collectFiles(sub, include, recursive));
1457
- }
1458
- return files;
1459
- }
1460
-
1461
- // src/contract/eval-reporting-suite.ts
1462
- var ANALYSIS_ARTIFACT2 = "analysis.json";
1245
+ const entries = await readdir(dir, { withFileTypes: true });
1246
+ const files = [];
1247
+ const subdirs = [];
1248
+ for (const entry of entries) {
1249
+ if (entry.isDirectory()) {
1250
+ if (recursive) subdirs.push(join(dir, entry.name));
1251
+ continue;
1252
+ }
1253
+ if (include(entry.name)) files.push(join(dir, entry.name));
1254
+ }
1255
+ files.sort();
1256
+ subdirs.sort();
1257
+ for (const sub of subdirs) files.push(...await collectFiles(sub, include, recursive));
1258
+ return files;
1259
+ }
1260
+ //#endregion
1261
+ //#region src/contract/eval-reporting-suite.ts
1262
+ /**
1263
+ * # `evalReportingSuite` — one call from runs (or a run dir) to `analysis.json`.
1264
+ *
1265
+ * A thin wrapper over the analysis primitive (`analyzeRuns`) and the on-disk
1266
+ * intake adapter (`fromRunRecordDir`). It does NOT reimplement any statistics,
1267
+ * distributions, or clustering — it resolves the input into validated
1268
+ * `RunRecord[]`, calls `analyzeRuns` with the options you'd pass it directly,
1269
+ * wraps the result in a small provenance envelope, and (optionally) writes a
1270
+ * single `analysis.json` artifact.
1271
+ *
1272
+ * ```ts
1273
+ * // From a directory of run files, write ./runs/analysis.json:
1274
+ * const suite = await evalReportingSuite('./runs', { write: true })
1275
+ * // From records already in memory, no write:
1276
+ * const suite = await evalReportingSuite(records, { analyze: { decisionThreshold: 0.03 } })
1277
+ * suite.report // the InsightReport — distributions, paired lift, findings rollup
1278
+ * ```
1279
+ */
1280
+ const ANALYSIS_ARTIFACT = "analysis.json";
1281
+ /**
1282
+ * Resolve runs (or a run dir/file), run `analyzeRuns`, and optionally persist a
1283
+ * single `analysis.json`. The only analysis logic lives in `analyzeRuns`; this
1284
+ * function is composition + I/O.
1285
+ */
1463
1286
  async function evalReportingSuite(input, options = {}) {
1464
- const fromPath = typeof input === "string";
1465
- let runs;
1466
- let files = [];
1467
- let rejected = [];
1468
- if (fromPath) {
1469
- const loaded = await fromRunRecordDir(input, options.load);
1470
- runs = loaded.runs;
1471
- files = loaded.files;
1472
- rejected = loaded.rejected;
1473
- } else {
1474
- runs = input;
1475
- }
1476
- if (runs.length === 0) {
1477
- throw new Error(
1478
- fromPath ? `evalReportingSuite: no RunRecords found at '${input}'` : "evalReportingSuite: no RunRecords to analyze"
1479
- );
1480
- }
1481
- const report = await analyzeRuns({ ...options.analyze, runs });
1482
- const result = {
1483
- report,
1484
- provenance: {
1485
- generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
1486
- runCount: runs.length,
1487
- sourcePath: fromPath ? input : null,
1488
- files,
1489
- rejected
1490
- },
1491
- writtenTo: null
1492
- };
1493
- const target = resolveWriteTarget(options.write, fromPath ? input : null);
1494
- if (target) {
1495
- await mkdir(dirname(target), { recursive: true });
1496
- await writeFile(target, `${JSON.stringify(result, null, 2)}
1497
- `, "utf8");
1498
- result.writtenTo = target;
1499
- }
1500
- return result;
1501
- }
1287
+ const fromPath = typeof input === "string";
1288
+ let runs;
1289
+ let files = [];
1290
+ let rejected = [];
1291
+ if (fromPath) {
1292
+ const loaded = await fromRunRecordDir(input, options.load);
1293
+ runs = loaded.runs;
1294
+ files = loaded.files;
1295
+ rejected = loaded.rejected;
1296
+ } else runs = input;
1297
+ if (runs.length === 0) throw new Error(fromPath ? `evalReportingSuite: no RunRecords found at '${input}'` : "evalReportingSuite: no RunRecords to analyze");
1298
+ const result = {
1299
+ report: await analyzeRuns({
1300
+ ...options.analyze,
1301
+ runs
1302
+ }),
1303
+ provenance: {
1304
+ generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
1305
+ runCount: runs.length,
1306
+ sourcePath: fromPath ? input : null,
1307
+ files,
1308
+ rejected
1309
+ },
1310
+ writtenTo: null
1311
+ };
1312
+ const target = resolveWriteTarget(options.write, fromPath ? input : null);
1313
+ if (target) {
1314
+ await mkdir(dirname(target), { recursive: true });
1315
+ await writeFile(target, `${JSON.stringify(result, null, 2)}\n`, "utf8");
1316
+ result.writtenTo = target;
1317
+ }
1318
+ return result;
1319
+ }
1320
+ /** Resolve where (if anywhere) to write `analysis.json`. Returns null when
1321
+ * writing is disabled. Throws on `write: true` with in-memory input — there is
1322
+ * no directory to anchor the artifact to, and silently inventing `cwd` would
1323
+ * scatter files. */
1502
1324
  function resolveWriteTarget(write, sourcePath) {
1503
- if (!write) return null;
1504
- if (typeof write === "string") {
1505
- const looksLikeDir = write.endsWith("/") || !write.endsWith(".json") && !write.endsWith(".jsonl");
1506
- return looksLikeDir ? join2(write, ANALYSIS_ARTIFACT2) : write;
1507
- }
1508
- if (sourcePath === null) {
1509
- throw new Error(
1510
- "evalReportingSuite: write:true needs a source path to anchor analysis.json \u2014 pass an explicit output path when analyzing in-memory records"
1511
- );
1512
- }
1513
- const isFile = sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl");
1514
- return isFile ? join2(dirname(sourcePath), ANALYSIS_ARTIFACT2) : join2(sourcePath, ANALYSIS_ARTIFACT2);
1325
+ if (!write) return null;
1326
+ if (typeof write === "string") return write.endsWith("/") || !write.endsWith(".json") && !write.endsWith(".jsonl") ? join(write, ANALYSIS_ARTIFACT) : write;
1327
+ if (sourcePath === null) throw new Error("evalReportingSuite: write:true needs a source path to anchor analysis.json pass an explicit output path when analyzing in-memory records");
1328
+ return sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl") ? join(dirname(sourcePath), ANALYSIS_ARTIFACT) : join(sourcePath, ANALYSIS_ARTIFACT);
1515
1329
  }
1516
-
1517
- // src/contract/diff.ts
1330
+ //#endregion
1331
+ //#region src/contract/diff.ts
1518
1332
  function keyForCell(cell) {
1519
- return JSON.stringify([cell.scenarioId, cell.rep]);
1333
+ return JSON.stringify([cell.scenarioId, cell.rep]);
1520
1334
  }
1335
+ /** Build the per-dimension delta map for a matched cell. Each judge name +
1336
+ * dimension name encountered on EITHER side appears in the result. */
1521
1337
  function diffDimensions(before, after) {
1522
- const out = {};
1523
- const judges = /* @__PURE__ */ new Set([...Object.keys(before), ...Object.keys(after)]);
1524
- for (const judge of judges) {
1525
- const beforeDims = before[judge] ?? {};
1526
- const afterDims = after[judge] ?? {};
1527
- const dims = /* @__PURE__ */ new Set([...Object.keys(beforeDims), ...Object.keys(afterDims)]);
1528
- const judgeOut = {};
1529
- for (const dim of dims) {
1530
- const rawBefore = beforeDims[dim];
1531
- const rawAfter = afterDims[dim];
1532
- const b = typeof rawBefore === "number" && Number.isFinite(rawBefore) ? rawBefore : null;
1533
- const a = typeof rawAfter === "number" && Number.isFinite(rawAfter) ? rawAfter : null;
1534
- judgeOut[dim] = {
1535
- before: b,
1536
- after: a,
1537
- delta: b !== null && a !== null ? a - b : null
1538
- };
1539
- }
1540
- out[judge] = judgeOut;
1541
- }
1542
- return out;
1543
- }
1338
+ const out = {};
1339
+ const judges = /* @__PURE__ */ new Set([...Object.keys(before), ...Object.keys(after)]);
1340
+ for (const judge of judges) {
1341
+ const beforeDims = before[judge] ?? {};
1342
+ const afterDims = after[judge] ?? {};
1343
+ const dims = /* @__PURE__ */ new Set([...Object.keys(beforeDims), ...Object.keys(afterDims)]);
1344
+ const judgeOut = {};
1345
+ for (const dim of dims) {
1346
+ const rawBefore = beforeDims[dim];
1347
+ const rawAfter = afterDims[dim];
1348
+ const b = typeof rawBefore === "number" && Number.isFinite(rawBefore) ? rawBefore : null;
1349
+ const a = typeof rawAfter === "number" && Number.isFinite(rawAfter) ? rawAfter : null;
1350
+ judgeOut[dim] = {
1351
+ before: b,
1352
+ after: a,
1353
+ delta: b !== null && a !== null ? a - b : null
1354
+ };
1355
+ }
1356
+ out[judge] = judgeOut;
1357
+ }
1358
+ return out;
1359
+ }
1360
+ /**
1361
+ * Diff two generation snapshots. Cells are matched on `(scenarioId, rep)`;
1362
+ * unmatched cells surface in `added` / `removed`. Aggregate fields are
1363
+ * recomputed from the snapshot's stored fields, not re-derived from cells —
1364
+ * this keeps the diff consistent with whatever aggregation the substrate
1365
+ * actually reported.
1366
+ */
1544
1367
  function diffGenerations(before, after) {
1545
- const beforeMap = new Map(before.cells.map((c) => [keyForCell(c), c]));
1546
- const afterMap = new Map(after.cells.map((c) => [keyForCell(c), c]));
1547
- const matched = [];
1548
- const removed = [];
1549
- const added = [];
1550
- for (const [key, beforeCell] of beforeMap) {
1551
- const afterCell = afterMap.get(key);
1552
- if (!afterCell) {
1553
- removed.push(beforeCell);
1554
- continue;
1555
- }
1556
- matched.push({
1557
- scenarioId: beforeCell.scenarioId,
1558
- rep: beforeCell.rep,
1559
- compositeBefore: beforeCell.compositeMean,
1560
- compositeAfter: afterCell.compositeMean,
1561
- compositeDelta: beforeCell.compositeMean === null || afterCell.compositeMean === null ? null : afterCell.compositeMean - beforeCell.compositeMean,
1562
- dimensions: diffDimensions(beforeCell.dimensions, afterCell.dimensions)
1563
- });
1564
- }
1565
- for (const [key, afterCell] of afterMap) {
1566
- if (!beforeMap.has(key)) added.push(afterCell);
1567
- }
1568
- return {
1569
- beforeIndex: before.index,
1570
- afterIndex: after.index,
1571
- beforeSurfaceHash: before.surfaceHash,
1572
- afterSurfaceHash: after.surfaceHash,
1573
- surfaceChanged: before.surfaceHash !== after.surfaceHash,
1574
- matched,
1575
- removed,
1576
- added,
1577
- compositeBefore: before.compositeMean,
1578
- compositeAfter: after.compositeMean,
1579
- compositeDelta: before.compositeMean === null || after.compositeMean === null ? null : after.compositeMean - before.compositeMean,
1580
- costUsdBefore: before.costUsd,
1581
- costUsdAfter: after.costUsd,
1582
- costUsdDelta: after.costUsd - before.costUsd,
1583
- durationMsBefore: before.durationMs,
1584
- durationMsAfter: after.durationMs,
1585
- durationMsDelta: after.durationMs - before.durationMs
1586
- };
1587
- }
1368
+ const beforeMap = new Map(before.cells.map((c) => [keyForCell(c), c]));
1369
+ const afterMap = new Map(after.cells.map((c) => [keyForCell(c), c]));
1370
+ const matched = [];
1371
+ const removed = [];
1372
+ const added = [];
1373
+ for (const [key, beforeCell] of beforeMap) {
1374
+ const afterCell = afterMap.get(key);
1375
+ if (!afterCell) {
1376
+ removed.push(beforeCell);
1377
+ continue;
1378
+ }
1379
+ matched.push({
1380
+ scenarioId: beforeCell.scenarioId,
1381
+ rep: beforeCell.rep,
1382
+ compositeBefore: beforeCell.compositeMean,
1383
+ compositeAfter: afterCell.compositeMean,
1384
+ compositeDelta: beforeCell.compositeMean === null || afterCell.compositeMean === null ? null : afterCell.compositeMean - beforeCell.compositeMean,
1385
+ dimensions: diffDimensions(beforeCell.dimensions, afterCell.dimensions)
1386
+ });
1387
+ }
1388
+ for (const [key, afterCell] of afterMap) if (!beforeMap.has(key)) added.push(afterCell);
1389
+ return {
1390
+ beforeIndex: before.index,
1391
+ afterIndex: after.index,
1392
+ beforeSurfaceHash: before.surfaceHash,
1393
+ afterSurfaceHash: after.surfaceHash,
1394
+ surfaceChanged: before.surfaceHash !== after.surfaceHash,
1395
+ matched,
1396
+ removed,
1397
+ added,
1398
+ compositeBefore: before.compositeMean,
1399
+ compositeAfter: after.compositeMean,
1400
+ compositeDelta: before.compositeMean === null || after.compositeMean === null ? null : after.compositeMean - before.compositeMean,
1401
+ costUsdBefore: before.costUsd,
1402
+ costUsdAfter: after.costUsd,
1403
+ costUsdDelta: after.costUsd - before.costUsd,
1404
+ durationMsBefore: before.durationMs,
1405
+ durationMsAfter: after.durationMs,
1406
+ durationMsDelta: after.durationMs - before.durationMs
1407
+ };
1408
+ }
1409
+ /** Highest-index generation, or null if the run recorded none. */
1588
1410
  function winnerOf(run) {
1589
- if (run.generations.length === 0) return null;
1590
- let winner = run.generations[0];
1591
- for (const gen of run.generations) {
1592
- if (gen.index > winner.index) winner = gen;
1593
- }
1594
- return winner;
1595
- }
1411
+ if (run.generations.length === 0) return null;
1412
+ let winner = run.generations[0];
1413
+ for (const gen of run.generations) if (gen.index > winner.index) winner = gen;
1414
+ return winner;
1415
+ }
1416
+ /**
1417
+ * Diff two full eval-runs. Produces baseline-vs-baseline and
1418
+ * winner-vs-winner generation diffs when both sides expose them, plus
1419
+ * run-level cost / lift / gate-decision deltas.
1420
+ */
1596
1421
  function diffRuns(before, after) {
1597
- const beforeWinner = winnerOf(before);
1598
- const afterWinner = winnerOf(after);
1599
- const baselineDiff = before.baseline && after.baseline ? diffGenerations(before.baseline, after.baseline) : null;
1600
- const winnersDiff = beforeWinner && afterWinner ? diffGenerations(beforeWinner, afterWinner) : null;
1601
- const beforeLift = before.holdoutLift ?? null;
1602
- const afterLift = after.holdoutLift ?? null;
1603
- return {
1604
- beforeRunId: before.runId,
1605
- afterRunId: after.runId,
1606
- beforeTimestamp: before.timestamp,
1607
- afterTimestamp: after.timestamp,
1608
- beforeGateDecision: before.gateDecision ?? null,
1609
- afterGateDecision: after.gateDecision ?? null,
1610
- beforeHoldoutLift: beforeLift,
1611
- afterHoldoutLift: afterLift,
1612
- holdoutLiftDelta: beforeLift !== null && afterLift !== null ? afterLift - beforeLift : null,
1613
- beforeTotalCostUsd: before.totalCostUsd,
1614
- afterTotalCostUsd: after.totalCostUsd,
1615
- totalCostUsdDelta: after.totalCostUsd - before.totalCostUsd,
1616
- beforeTotalDurationMs: before.totalDurationMs,
1617
- afterTotalDurationMs: after.totalDurationMs,
1618
- totalDurationMsDelta: after.totalDurationMs - before.totalDurationMs,
1619
- baselineDiff,
1620
- winnersDiff
1621
- };
1622
- }
1422
+ const beforeWinner = winnerOf(before);
1423
+ const afterWinner = winnerOf(after);
1424
+ const baselineDiff = before.baseline && after.baseline ? diffGenerations(before.baseline, after.baseline) : null;
1425
+ const winnersDiff = beforeWinner && afterWinner ? diffGenerations(beforeWinner, afterWinner) : null;
1426
+ const beforeLift = before.holdoutLift ?? null;
1427
+ const afterLift = after.holdoutLift ?? null;
1428
+ return {
1429
+ beforeRunId: before.runId,
1430
+ afterRunId: after.runId,
1431
+ beforeTimestamp: before.timestamp,
1432
+ afterTimestamp: after.timestamp,
1433
+ beforeGateDecision: before.gateDecision ?? null,
1434
+ afterGateDecision: after.gateDecision ?? null,
1435
+ beforeHoldoutLift: beforeLift,
1436
+ afterHoldoutLift: afterLift,
1437
+ holdoutLiftDelta: beforeLift !== null && afterLift !== null ? afterLift - beforeLift : null,
1438
+ beforeTotalCostUsd: before.totalCostUsd,
1439
+ afterTotalCostUsd: after.totalCostUsd,
1440
+ totalCostUsdDelta: after.totalCostUsd - before.totalCostUsd,
1441
+ beforeTotalDurationMs: before.totalDurationMs,
1442
+ afterTotalDurationMs: after.totalDurationMs,
1443
+ totalDurationMsDelta: after.totalDurationMs - before.totalDurationMs,
1444
+ baselineDiff,
1445
+ winnersDiff
1446
+ };
1447
+ }
1448
+ /**
1449
+ * Within-run baseline → winning-generation diff. The natural "what did the
1450
+ * improvement loop produce" view for a single run. Returns null when the
1451
+ * run never reached a generation past baseline (errored early, or the gate
1452
+ * shipped the baseline as-is).
1453
+ */
1623
1454
  function diffRunBaselineToWinner(run) {
1624
- if (!run.baseline) return null;
1625
- const winner = winnerOf(run);
1626
- if (!winner || winner.index === run.baseline.index) return null;
1627
- return diffGenerations(run.baseline, winner);
1455
+ if (!run.baseline) return null;
1456
+ const winner = winnerOf(run);
1457
+ if (!winner || winner.index === run.baseline.index) return null;
1458
+ return diffGenerations(run.baseline, winner);
1628
1459
  }
1629
-
1630
- // src/contract/intake/agent-trace.ts
1460
+ //#endregion
1461
+ //#region src/contract/intake/agent-trace.ts
1631
1462
  function rangeLines(r) {
1632
- return Math.max(0, r.end_line - r.start_line + 1);
1463
+ return Math.max(0, r.end_line - r.start_line + 1);
1633
1464
  }
1465
+ /**
1466
+ * Build a commit → provenance index from Agent Trace records. Multiple records
1467
+ * for the same revision are merged. Records without `vcs.revision` are skipped
1468
+ * (the SHA is the join key — without it there is nothing to correlate against).
1469
+ */
1634
1470
  function parseAgentTrace(records) {
1635
- const acc = /* @__PURE__ */ new Map();
1636
- for (const record of records) {
1637
- const sha = record.vcs?.revision;
1638
- if (!sha) continue;
1639
- let a = acc.get(sha);
1640
- if (!a) {
1641
- a = {
1642
- models: /* @__PURE__ */ new Set(),
1643
- tools: /* @__PURE__ */ new Set(),
1644
- files: /* @__PURE__ */ new Set(),
1645
- conversationCount: 0,
1646
- lineCount: 0,
1647
- humanInvolved: false
1648
- };
1649
- acc.set(sha, a);
1650
- }
1651
- if (record.tool?.name) a.tools.add(record.tool.name);
1652
- for (const file of record.files ?? []) {
1653
- a.files.add(file.path);
1654
- for (const conv of file.conversations ?? []) {
1655
- a.conversationCount += 1;
1656
- for (const range of conv.ranges ?? []) {
1657
- const contributor = range.contributor ?? conv.contributor;
1658
- a.lineCount += rangeLines(range);
1659
- if (!contributor) continue;
1660
- if (contributor.type === "human" || contributor.type === "mixed") {
1661
- a.humanInvolved = true;
1662
- }
1663
- if ((contributor.type === "ai" || contributor.type === "mixed") && contributor.model_id) {
1664
- a.models.add(contributor.model_id);
1665
- }
1666
- }
1667
- }
1668
- }
1669
- }
1670
- const index = /* @__PURE__ */ new Map();
1671
- for (const [sha, a] of acc) {
1672
- index.set(sha, {
1673
- commitSha: sha,
1674
- aiModels: [...a.models].sort(),
1675
- tools: [...a.tools].sort(),
1676
- conversationCount: a.conversationCount,
1677
- fileCount: a.files.size,
1678
- lineCount: a.lineCount,
1679
- humanInvolved: a.humanInvolved
1680
- });
1681
- }
1682
- return index;
1683
- }
1471
+ const acc = /* @__PURE__ */ new Map();
1472
+ for (const record of records) {
1473
+ const sha = record.vcs?.revision;
1474
+ if (!sha) continue;
1475
+ let a = acc.get(sha);
1476
+ if (!a) {
1477
+ a = {
1478
+ models: /* @__PURE__ */ new Set(),
1479
+ tools: /* @__PURE__ */ new Set(),
1480
+ files: /* @__PURE__ */ new Set(),
1481
+ conversationCount: 0,
1482
+ lineCount: 0,
1483
+ humanInvolved: false
1484
+ };
1485
+ acc.set(sha, a);
1486
+ }
1487
+ if (record.tool?.name) a.tools.add(record.tool.name);
1488
+ for (const file of record.files ?? []) {
1489
+ a.files.add(file.path);
1490
+ for (const conv of file.conversations ?? []) {
1491
+ a.conversationCount += 1;
1492
+ for (const range of conv.ranges ?? []) {
1493
+ const contributor = range.contributor ?? conv.contributor;
1494
+ a.lineCount += rangeLines(range);
1495
+ if (!contributor) continue;
1496
+ if (contributor.type === "human" || contributor.type === "mixed") a.humanInvolved = true;
1497
+ if ((contributor.type === "ai" || contributor.type === "mixed") && contributor.model_id) a.models.add(contributor.model_id);
1498
+ }
1499
+ }
1500
+ }
1501
+ }
1502
+ const index = /* @__PURE__ */ new Map();
1503
+ for (const [sha, a] of acc) index.set(sha, {
1504
+ commitSha: sha,
1505
+ aiModels: [...a.models].sort(),
1506
+ tools: [...a.tools].sort(),
1507
+ conversationCount: a.conversationCount,
1508
+ fileCount: a.files.size,
1509
+ lineCount: a.lineCount,
1510
+ humanInvolved: a.humanInvolved
1511
+ });
1512
+ return index;
1513
+ }
1514
+ /**
1515
+ * Partition runs by the AI model(s) that authored the code at each run's
1516
+ * `commitSha`. Feed `byModel.get(modelId)` to `analyzeRuns`, or compare two
1517
+ * model cohorts via `analyzeRuns({ runs: a, baselineRuns: b })` for a lift CI
1518
+ * on "model A's code vs model B's code".
1519
+ */
1684
1520
  function partitionRunsByAuthoringModel(runs, index) {
1685
- const byModel = /* @__PURE__ */ new Map();
1686
- const unattributed = [];
1687
- for (const run of runs) {
1688
- const provenance = index.get(run.commitSha);
1689
- if (!provenance || provenance.aiModels.length === 0) {
1690
- unattributed.push(run);
1691
- continue;
1692
- }
1693
- for (const model of provenance.aiModels) {
1694
- const cohort = byModel.get(model) ?? [];
1695
- cohort.push(run);
1696
- byModel.set(model, cohort);
1697
- }
1698
- }
1699
- return { byModel, unattributed };
1700
- }
1701
-
1702
- // src/contract/intake/feedback-table.ts
1521
+ const byModel = /* @__PURE__ */ new Map();
1522
+ const unattributed = [];
1523
+ for (const run of runs) {
1524
+ const provenance = index.get(run.commitSha);
1525
+ if (!provenance || provenance.aiModels.length === 0) {
1526
+ unattributed.push(run);
1527
+ continue;
1528
+ }
1529
+ for (const model of provenance.aiModels) {
1530
+ const cohort = byModel.get(model) ?? [];
1531
+ cohort.push(run);
1532
+ byModel.set(model, cohort);
1533
+ }
1534
+ }
1535
+ return {
1536
+ byModel,
1537
+ unattributed
1538
+ };
1539
+ }
1540
+ //#endregion
1541
+ //#region src/contract/intake/feedback-table.ts
1703
1542
  function fromFeedbackTable(opts) {
1704
- const { ratings, meta = [], scale, emitRaterScores = true } = opts;
1705
- const metaByRun = new Map(meta.map((m) => [m.runId, m]));
1706
- const normalise = (rating) => {
1707
- if (typeof rating === "boolean") return rating ? 1 : 0;
1708
- if (!Number.isFinite(rating)) return Number.NaN;
1709
- if (!scale) return rating;
1710
- const { min, max } = scale;
1711
- if (max === min) return rating;
1712
- return (rating - min) / (max - min);
1713
- };
1714
- const byRun = /* @__PURE__ */ new Map();
1715
- for (const row of ratings) {
1716
- const list = byRun.get(row.runId) ?? [];
1717
- list.push(row);
1718
- byRun.set(row.runId, list);
1719
- }
1720
- const runs = [];
1721
- const raterScores = [];
1722
- for (const [runId, rowsForRun] of byRun) {
1723
- const normalised = rowsForRun.map((r) => ({ rater: r.rater, score: normalise(r.rating) })).filter((r) => Number.isFinite(r.score));
1724
- if (normalised.length === 0) continue;
1725
- const meanScore = normalised.reduce((s, r) => s + r.score, 0) / normalised.length;
1726
- const runMeta = metaByRun.get(runId) ?? { runId };
1727
- const judgeScores = {
1728
- perJudge: Object.fromEntries(normalised.map((r) => [r.rater, { rating: r.score }])),
1729
- perDimMean: { rating: meanScore },
1730
- composite: meanScore
1731
- };
1732
- const splitTag = runMeta.splitTag ?? "holdout";
1733
- const outcome = {
1734
- ...splitTag === "holdout" ? { holdoutScore: meanScore } : { searchScore: meanScore },
1735
- raw: Object.fromEntries(normalised.map((r) => [`rater:${r.rater}`, r.score])),
1736
- judgeScores
1737
- };
1738
- const costUsd = runMeta.costUsd ?? null;
1739
- runs.push({
1740
- runId,
1741
- experimentId: runMeta.experimentId ?? "feedback-corpus",
1742
- candidateId: runMeta.candidateId ?? runId,
1743
- seed: 0,
1744
- model: runMeta.model ?? "unknown@unknown",
1745
- promptHash: runMeta.promptHash ?? "sha256:unknown",
1746
- configHash: runMeta.configHash ?? "sha256:unknown",
1747
- commitSha: runMeta.commitSha ?? "unknown",
1748
- wallMs: runMeta.wallMs ?? 0,
1749
- costUsd,
1750
- costProvenance: costUsd === null ? { kind: "uncaptured", usd: null } : { kind: "observed", usd: costUsd },
1751
- tokenUsage: { input: 0, output: 0 },
1752
- terminalOutcome: "unknown",
1753
- outcome,
1754
- splitTag,
1755
- scenarioId: runMeta.scenarioId ?? runId
1756
- });
1757
- if (emitRaterScores) {
1758
- for (const r of normalised) raterScores.push({ runId, rater: r.rater, score: r.score });
1759
- }
1760
- }
1761
- return { runs, raterScores };
1762
- }
1763
-
1764
- // src/contract/intake/otel-spans.ts
1765
- var TASK_SCORE_ATTR_KEYS = [
1766
- "gen_ai.evaluation.score.value",
1767
- "tangle.task.score",
1768
- "eval.score",
1769
- "tangle.score"
1543
+ const { ratings, meta = [], scale, emitRaterScores = true } = opts;
1544
+ const metaByRun = new Map(meta.map((m) => [m.runId, m]));
1545
+ const normalise = (rating) => {
1546
+ if (typeof rating === "boolean") return rating ? 1 : 0;
1547
+ if (!Number.isFinite(rating)) return NaN;
1548
+ if (!scale) return rating;
1549
+ const { min, max } = scale;
1550
+ if (max === min) return rating;
1551
+ return (rating - min) / (max - min);
1552
+ };
1553
+ const byRun = /* @__PURE__ */ new Map();
1554
+ for (const row of ratings) {
1555
+ const list = byRun.get(row.runId) ?? [];
1556
+ list.push(row);
1557
+ byRun.set(row.runId, list);
1558
+ }
1559
+ const runs = [];
1560
+ const raterScores = [];
1561
+ for (const [runId, rowsForRun] of byRun) {
1562
+ const normalised = rowsForRun.map((r) => ({
1563
+ rater: r.rater,
1564
+ score: normalise(r.rating)
1565
+ })).filter((r) => Number.isFinite(r.score));
1566
+ if (normalised.length === 0) continue;
1567
+ const meanScore = normalised.reduce((s, r) => s + r.score, 0) / normalised.length;
1568
+ const runMeta = metaByRun.get(runId) ?? { runId };
1569
+ const judgeScores = {
1570
+ perJudge: Object.fromEntries(normalised.map((r) => [r.rater, { rating: r.score }])),
1571
+ perDimMean: { rating: meanScore },
1572
+ composite: meanScore
1573
+ };
1574
+ const splitTag = runMeta.splitTag ?? "holdout";
1575
+ const outcome = {
1576
+ ...splitTag === "holdout" ? { holdoutScore: meanScore } : { searchScore: meanScore },
1577
+ raw: Object.fromEntries(normalised.map((r) => [`rater:${r.rater}`, r.score])),
1578
+ judgeScores
1579
+ };
1580
+ const costUsd = runMeta.costUsd ?? null;
1581
+ runs.push({
1582
+ runId,
1583
+ experimentId: runMeta.experimentId ?? "feedback-corpus",
1584
+ candidateId: runMeta.candidateId ?? runId,
1585
+ seed: 0,
1586
+ model: runMeta.model ?? "unknown@unknown",
1587
+ promptHash: runMeta.promptHash ?? "sha256:unknown",
1588
+ configHash: runMeta.configHash ?? "sha256:unknown",
1589
+ commitSha: runMeta.commitSha ?? "unknown",
1590
+ wallMs: runMeta.wallMs ?? 0,
1591
+ costUsd,
1592
+ costProvenance: costUsd === null ? {
1593
+ kind: "uncaptured",
1594
+ usd: null
1595
+ } : {
1596
+ kind: "observed",
1597
+ usd: costUsd
1598
+ },
1599
+ tokenUsage: {
1600
+ input: 0,
1601
+ output: 0
1602
+ },
1603
+ terminalOutcome: "unknown",
1604
+ outcome,
1605
+ splitTag,
1606
+ scenarioId: runMeta.scenarioId ?? runId
1607
+ });
1608
+ if (emitRaterScores) for (const r of normalised) raterScores.push({
1609
+ runId,
1610
+ rater: r.rater,
1611
+ score: r.score
1612
+ });
1613
+ }
1614
+ return {
1615
+ runs,
1616
+ raterScores
1617
+ };
1618
+ }
1619
+ //#endregion
1620
+ //#region src/contract/intake/otel-spans.ts
1621
+ /**
1622
+ * # `intake/otel-spans` — OTel `TraceSpanEvent[]` → `RunRecord[]`.
1623
+ *
1624
+ * Turns an existing observability stream into the substrate-canonical
1625
+ * `RunRecord` shape so consumers with logs but no eval discipline can
1626
+ * call `analyzeRuns()` against their production traffic immediately.
1627
+ *
1628
+ * Pivot rule: spans are grouped by `tangle.runId` (the same attribute the
1629
+ * hosted-tier wire format uses) or, when absent, by `traceId`. One group
1630
+ * becomes one `RunRecord`. The root span (no `parentSpanId`) supplies:
1631
+ *
1632
+ * - `runId` (the group key)
1633
+ * - `wallMs` from `endTimeUnixNano - startTimeUnixNano`
1634
+ * - `model` from `gen_ai.request.model` / `llm.model` / `tangle.model`
1635
+ * - task failure class and detail from explicit `tangle.task.*` attributes
1636
+ * - cost from `cost.usd` / `gen_ai.usage.cost_usd` / `tangle.cost.usd`
1637
+ * - token usage from model-call input, output, cache-read, and cache-write
1638
+ * attributes without double-counting aggregate parent spans
1639
+ * - task quality from an explicit `scoreForRun` callback or a designated
1640
+ * evaluation attribute on a root / `EVALUATOR` span; `outcome.raw`
1641
+ * collects every numeric attribute without promoting it to task quality.
1642
+ *
1643
+ * Errored tool, model, and child-agent spans contribute to execution-error
1644
+ * counts. Root process, guardrail, evaluator, propagated parent, and unknown
1645
+ * errors retain separate counters. Only one failed root can set
1646
+ * `RunRecord.terminalOutcome` and `RunRecord.terminalFailureReason`; a child
1647
+ * error cannot become a task failure.
1648
+ */
1649
+ const TASK_SCORE_ATTR_KEYS = [
1650
+ "gen_ai.evaluation.score.value",
1651
+ "tangle.task.score",
1652
+ "eval.score",
1653
+ "tangle.score"
1770
1654
  ];
1771
- var MODEL_KEYS = ["tangle.model", ...LLM_MODEL_ATTR_KEYS, "model"];
1772
- var PROMPT_HASH_KEYS = ["tangle.prompt_hash", "prompt.hash"];
1773
- var CONFIG_HASH_KEYS = ["tangle.config_hash", "config.hash"];
1655
+ const MODEL_KEYS = [
1656
+ "tangle.model",
1657
+ ...LLM_MODEL_ATTR_KEYS,
1658
+ "model"
1659
+ ];
1660
+ const PROMPT_HASH_KEYS = ["tangle.prompt_hash", "prompt.hash"];
1661
+ const CONFIG_HASH_KEYS = ["tangle.config_hash", "config.hash"];
1774
1662
  function fromOtelSpans(opts) {
1775
- const { spans, defaultSplit = "holdout", experimentId = "otel-corpus" } = opts;
1776
- const grouped = groupSpans(spans);
1777
- const runs = [];
1778
- for (const [groupKey, groupSpans2] of grouped) {
1779
- const root = findRoot(groupSpans2);
1780
- if (!root) continue;
1781
- const measurements = summarizeExecutionMeasurements(
1782
- groupSpans2.map((span) => ({
1783
- id: span.spanId,
1784
- ...span.parentSpanId ? { parentId: span.parentSpanId } : {},
1785
- attributes: span.attributes,
1786
- modelCall: isExplicitModelCall(span),
1787
- aggregate: isExplicitAggregate(span)
1788
- }))
1789
- );
1790
- const callSpanIds = new Set(measurements.callSpanIds);
1791
- const callSpans = groupSpans2.filter((span) => callSpanIds.has(span.spanId));
1792
- const wallMs = unixNanoDurationMs(root.startTimeUnixNano, root.endTimeUnixNano);
1793
- const model = readAttrString(callSpans, MODEL_KEYS) ?? readAttrString(groupSpans2, MODEL_KEYS) ?? "unknown@unknown";
1794
- const capturedCost = (measurements.cost.complete ? measurements.cost.value : void 0) ?? measurements.aggregate?.costUsd;
1795
- const costUsd = capturedCost ?? null;
1796
- const scenarioId = readConsistentScenarioId(groupKey, groupSpans2) ?? groupKey;
1797
- const promptHash = readAttrString(groupSpans2, PROMPT_HASH_KEYS) ?? "sha256:unknown";
1798
- const configHash = readAttrString(groupSpans2, CONFIG_HASH_KEYS) ?? "sha256:unknown";
1799
- const score = resolveTaskScore(groupKey, groupSpans2, opts.scoreForRun);
1800
- const taskFailure = readTaskFailureLabels(
1801
- groupSpans2.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span)),
1802
- `fromOtelSpans: run '${groupKey}'`
1803
- );
1804
- const rawNumeric = collectNumericAttrs(groupSpans2);
1805
- const errorSummary = summarizeTraceErrors(
1806
- groupSpans2.map((span) => ({
1807
- id: spanIdentity(span),
1808
- ...span.parentSpanId ? { parentId: parentIdentity(span) } : {},
1809
- role: errorRoleForSpan(span),
1810
- error: span.status?.code === "ERROR",
1811
- processRoot: !span.parentSpanId && isTerminalRootCandidate(span)
1812
- }))
1813
- );
1814
- rawNumeric.error_span_count = errorSummary.total;
1815
- rawNumeric.execution_error_count = errorSummary.execution;
1816
- rawNumeric.process_error_count = errorSummary.process;
1817
- rawNumeric.guardrail_error_count = errorSummary.guardrail;
1818
- rawNumeric.judge_error_count = errorSummary.evaluation;
1819
- rawNumeric.propagated_error_count = errorSummary.propagated;
1820
- rawNumeric.unclassified_error_count = errorSummary.unclassified;
1821
- rawNumeric.llm_span_count = measurements.modelCallCount;
1822
- if (measurements.cost.value !== void 0 && !measurements.cost.complete) {
1823
- rawNumeric.partial_observed_cost_usd = measurements.cost.value;
1824
- }
1825
- recordAggregateMeasurements(rawNumeric, measurements.aggregate);
1826
- const judgeScores = score !== void 0 ? {
1827
- perJudge: { "otel-derived": { score } },
1828
- perDimMean: { score },
1829
- composite: score
1830
- } : void 0;
1831
- const terminalOutcome = terminalOutcomeFromRoots(groupSpans2);
1832
- const failedRoot = terminalOutcome === "failed" ? groupSpans2.find(
1833
- (span) => !span.parentSpanId && isTerminalRootCandidate(span) && span.status?.code === "ERROR"
1834
- ) : void 0;
1835
- const outcome = {
1836
- raw: rawNumeric,
1837
- ...judgeScores ? { judgeScores } : {}
1838
- };
1839
- if (score !== void 0) {
1840
- if (defaultSplit === "holdout") outcome.holdoutScore = score;
1841
- else outcome.searchScore = score;
1842
- }
1843
- runs.push({
1844
- runId: groupKey,
1845
- experimentId,
1846
- candidateId: root.attributes["tangle.candidateId"] ?? "otel-default",
1847
- seed: 0,
1848
- model,
1849
- promptHash,
1850
- configHash,
1851
- commitSha: root.attributes["tangle.commit_sha"] ?? "unknown",
1852
- wallMs,
1853
- costUsd,
1854
- costProvenance: capturedCost === void 0 ? { kind: "uncaptured", usd: null } : { kind: "observed", usd: capturedCost },
1855
- tokenUsage: measurements.tokenUsage,
1856
- terminalOutcome,
1857
- ...failedRoot ? { terminalFailureReason: failedRoot.status?.message ?? failedRoot.name } : {},
1858
- outcome,
1859
- ...taskFailure,
1860
- splitTag: defaultSplit,
1861
- scenarioId
1862
- });
1863
- }
1864
- return runs;
1663
+ const { spans, defaultSplit = "holdout", experimentId = "otel-corpus" } = opts;
1664
+ const grouped = groupSpans(spans);
1665
+ const runs = [];
1666
+ for (const [groupKey, groupSpans] of grouped) {
1667
+ const root = findRoot(groupSpans);
1668
+ if (!root) continue;
1669
+ const measurements = summarizeExecutionMeasurements(groupSpans.map((span) => ({
1670
+ id: span.spanId,
1671
+ ...span.parentSpanId ? { parentId: span.parentSpanId } : {},
1672
+ attributes: span.attributes,
1673
+ modelCall: isExplicitModelCall(span),
1674
+ aggregate: isExplicitAggregate(span)
1675
+ })));
1676
+ const callSpanIds = new Set(measurements.callSpanIds);
1677
+ const callSpans = groupSpans.filter((span) => callSpanIds.has(span.spanId));
1678
+ const wallMs = unixNanoDurationMs(root.startTimeUnixNano, root.endTimeUnixNano);
1679
+ const model = readAttrString(callSpans, MODEL_KEYS) ?? readAttrString(groupSpans, MODEL_KEYS) ?? "unknown@unknown";
1680
+ const capturedCost = (measurements.cost.complete ? measurements.cost.value : void 0) ?? measurements.aggregate?.costUsd;
1681
+ const costUsd = capturedCost ?? null;
1682
+ const scenarioId = readConsistentScenarioId(groupKey, groupSpans) ?? groupKey;
1683
+ const promptHash = readAttrString(groupSpans, PROMPT_HASH_KEYS) ?? "sha256:unknown";
1684
+ const configHash = readAttrString(groupSpans, CONFIG_HASH_KEYS) ?? "sha256:unknown";
1685
+ const score = resolveTaskScore(groupKey, groupSpans, opts.scoreForRun);
1686
+ const taskFailure = readTaskFailureLabels(groupSpans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span)), `fromOtelSpans: run '${groupKey}'`);
1687
+ const rawNumeric = collectNumericAttrs(groupSpans);
1688
+ const errorSummary = summarizeTraceErrors(groupSpans.map((span) => ({
1689
+ id: spanIdentity(span),
1690
+ ...span.parentSpanId ? { parentId: parentIdentity(span) } : {},
1691
+ role: errorRoleForSpan(span),
1692
+ error: span.status?.code === "ERROR",
1693
+ processRoot: !span.parentSpanId && isTerminalRootCandidate(span)
1694
+ })));
1695
+ rawNumeric.error_span_count = errorSummary.total;
1696
+ rawNumeric.execution_error_count = errorSummary.execution;
1697
+ rawNumeric.process_error_count = errorSummary.process;
1698
+ rawNumeric.guardrail_error_count = errorSummary.guardrail;
1699
+ rawNumeric.judge_error_count = errorSummary.evaluation;
1700
+ rawNumeric.propagated_error_count = errorSummary.propagated;
1701
+ rawNumeric.unclassified_error_count = errorSummary.unclassified;
1702
+ rawNumeric.llm_span_count = measurements.modelCallCount;
1703
+ if (measurements.cost.value !== void 0 && !measurements.cost.complete) rawNumeric.partial_observed_cost_usd = measurements.cost.value;
1704
+ recordAggregateMeasurements(rawNumeric, measurements.aggregate);
1705
+ const judgeScores = score !== void 0 ? {
1706
+ perJudge: { "otel-derived": { score } },
1707
+ perDimMean: { score },
1708
+ composite: score
1709
+ } : void 0;
1710
+ const terminalOutcome = terminalOutcomeFromRoots(groupSpans);
1711
+ const failedRoot = terminalOutcome === "failed" ? groupSpans.find((span) => !span.parentSpanId && isTerminalRootCandidate(span) && span.status?.code === "ERROR") : void 0;
1712
+ const outcome = {
1713
+ raw: rawNumeric,
1714
+ ...judgeScores ? { judgeScores } : {}
1715
+ };
1716
+ if (score !== void 0) if (defaultSplit === "holdout") outcome.holdoutScore = score;
1717
+ else outcome.searchScore = score;
1718
+ runs.push({
1719
+ runId: groupKey,
1720
+ experimentId,
1721
+ candidateId: root.attributes["tangle.candidateId"] ?? "otel-default",
1722
+ seed: 0,
1723
+ model,
1724
+ promptHash,
1725
+ configHash,
1726
+ commitSha: root.attributes["tangle.commit_sha"] ?? "unknown",
1727
+ wallMs,
1728
+ costUsd,
1729
+ costProvenance: capturedCost === void 0 ? {
1730
+ kind: "uncaptured",
1731
+ usd: null
1732
+ } : {
1733
+ kind: "observed",
1734
+ usd: capturedCost
1735
+ },
1736
+ tokenUsage: measurements.tokenUsage,
1737
+ terminalOutcome,
1738
+ ...failedRoot ? { terminalFailureReason: failedRoot.status?.message ?? failedRoot.name } : {},
1739
+ outcome,
1740
+ ...taskFailure,
1741
+ splitTag: defaultSplit,
1742
+ scenarioId
1743
+ });
1744
+ }
1745
+ return runs;
1865
1746
  }
1866
1747
  function terminalOutcomeFromRoots(spans) {
1867
- const roots = spans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span));
1868
- if (roots.length !== 1) return "unknown";
1869
- if (roots[0].status?.code === "OK") return "succeeded";
1870
- if (roots[0].status?.code === "ERROR") return "failed";
1871
- return "unknown";
1748
+ const roots = spans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span));
1749
+ if (roots.length !== 1) return "unknown";
1750
+ if (roots[0].status?.code === "OK") return "succeeded";
1751
+ if (roots[0].status?.code === "ERROR") return "failed";
1752
+ return "unknown";
1872
1753
  }
1873
1754
  function isTerminalRootCandidate(span) {
1874
- const role = errorRoleForSpan(span);
1875
- if (role === "LLM" || role === "TOOL" || role === "GUARDRAIL" || role === "EVALUATOR") {
1876
- return false;
1877
- }
1878
- return true;
1755
+ const role = errorRoleForSpan(span);
1756
+ if (role === "LLM" || role === "TOOL" || role === "GUARDRAIL" || role === "EVALUATOR") return false;
1757
+ return true;
1879
1758
  }
1880
1759
  function readSpanKind(span) {
1881
- return readAttrString([span], [...SPAN_KIND_ATTR_KEYS, "span.kind"])?.toUpperCase();
1760
+ return readAttrString([span], [...SPAN_KIND_ATTR_KEYS, "span.kind"])?.toUpperCase();
1882
1761
  }
1883
1762
  function errorRoleForSpan(span) {
1884
- return classifyOtlpSpanRole({
1885
- kind: readSpanKind(span),
1886
- name: span.name,
1887
- attributes: span.attributes
1888
- });
1763
+ return classifyOtlpSpanRole({
1764
+ kind: readSpanKind(span),
1765
+ name: span.name,
1766
+ attributes: span.attributes
1767
+ });
1889
1768
  }
1890
1769
  function spanIdentity(span) {
1891
- return `${span.traceId}:${span.spanId}`;
1770
+ return `${span.traceId}:${span.spanId}`;
1892
1771
  }
1893
1772
  function parentIdentity(span) {
1894
- return `${span.traceId}:${span.parentSpanId}`;
1773
+ return `${span.traceId}:${span.parentSpanId}`;
1895
1774
  }
1896
1775
  function isExplicitModelCall(span) {
1897
- return isOtlpModelCall({
1898
- kind: readSpanKind(span),
1899
- name: span.name,
1900
- attributes: span.attributes
1901
- });
1776
+ return isOtlpModelCall({
1777
+ kind: readSpanKind(span),
1778
+ name: span.name,
1779
+ attributes: span.attributes
1780
+ });
1902
1781
  }
1903
1782
  function isExplicitAggregate(span) {
1904
- const kind = readSpanKind(span);
1905
- return kind !== void 0 && kind !== "LLM";
1783
+ const kind = readSpanKind(span);
1784
+ return kind !== void 0 && kind !== "LLM";
1906
1785
  }
1907
1786
  function groupSpans(spans) {
1908
- const m = /* @__PURE__ */ new Map();
1909
- for (const span of spans) {
1910
- const key = span["tangle.runId"] ?? span.traceId;
1911
- const list = m.get(key) ?? [];
1912
- list.push(span);
1913
- m.set(key, list);
1914
- }
1915
- return m;
1787
+ const m = /* @__PURE__ */ new Map();
1788
+ for (const span of spans) {
1789
+ const key = span["tangle.runId"] ?? span.traceId;
1790
+ const list = m.get(key) ?? [];
1791
+ list.push(span);
1792
+ m.set(key, list);
1793
+ }
1794
+ return m;
1916
1795
  }
1917
1796
  function findRoot(group) {
1918
- const structuralRoots = group.filter((span) => !span.parentSpanId);
1919
- const terminalRoots = structuralRoots.filter(isTerminalRootCandidate);
1920
- const pool = terminalRoots.length > 0 ? terminalRoots : structuralRoots.length > 0 ? structuralRoots : group;
1921
- return orderSpans(pool)[0];
1797
+ const structuralRoots = group.filter((span) => !span.parentSpanId);
1798
+ const terminalRoots = structuralRoots.filter(isTerminalRootCandidate);
1799
+ return orderSpans(terminalRoots.length > 0 ? terminalRoots : structuralRoots.length > 0 ? structuralRoots : group)[0];
1922
1800
  }
1923
1801
  function readAttrString(spans, keys) {
1924
- for (const span of spans) {
1925
- for (const key of keys) {
1926
- const v = span.attributes[key];
1927
- if (typeof v === "string" && v.length > 0) return v;
1928
- }
1929
- }
1930
- return void 0;
1802
+ for (const span of spans) for (const key of keys) {
1803
+ const v = span.attributes[key];
1804
+ if (typeof v === "string" && v.length > 0) return v;
1805
+ }
1931
1806
  }
1932
1807
  function readConsistentScenarioId(runId, spans) {
1933
- const values = /* @__PURE__ */ new Set();
1934
- for (const span of spans) {
1935
- const topLevel = span["tangle.scenarioId"];
1936
- if (typeof topLevel === "string" && topLevel.length > 0) values.add(topLevel);
1937
- const attribute = span.attributes["tangle.scenarioId"];
1938
- if (typeof attribute === "string" && attribute.length > 0) values.add(attribute);
1939
- }
1940
- if (values.size > 1) {
1941
- throw new ValidationError(
1942
- `fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`
1943
- );
1944
- }
1945
- return values.values().next().value;
1808
+ const values = /* @__PURE__ */ new Set();
1809
+ for (const span of spans) {
1810
+ const topLevel = span["tangle.scenarioId"];
1811
+ if (typeof topLevel === "string" && topLevel.length > 0) values.add(topLevel);
1812
+ const attribute = span.attributes["tangle.scenarioId"];
1813
+ if (typeof attribute === "string" && attribute.length > 0) values.add(attribute);
1814
+ }
1815
+ if (values.size > 1) throw new ValidationError(`fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`);
1816
+ return values.values().next().value;
1946
1817
  }
1947
1818
  function resolveTaskScore(runId, spans, scoreForRun) {
1948
- const orderedSpans = orderSpans(spans);
1949
- const sources = [];
1950
- if (scoreForRun) {
1951
- const supplied = scoreForRun(runId, orderedSpans);
1952
- if (supplied !== void 0) {
1953
- if (typeof supplied !== "number" || !Number.isFinite(supplied)) {
1954
- throw new ValidationError(
1955
- `fromOtelSpans: scoreForRun returned a non-finite number for run '${runId}'`
1956
- );
1957
- }
1958
- sources.push({ label: "scoreForRun", value: supplied });
1959
- }
1960
- }
1961
- for (const span of orderedSpans) {
1962
- const role = errorRoleForSpan(span);
1963
- if (span.parentSpanId && role !== "EVALUATOR") continue;
1964
- if (role === "EVALUATOR" && span.status?.code === "ERROR") continue;
1965
- for (const key of TASK_SCORE_ATTR_KEYS) {
1966
- if (!Object.hasOwn(span.attributes, key)) continue;
1967
- sources.push({
1968
- label: `span '${span.spanId}' attribute '${key}'`,
1969
- value: parseTaskScoreAttribute(runId, span.spanId, key, span.attributes[key])
1970
- });
1971
- }
1972
- }
1973
- if (sources.length === 0) return void 0;
1974
- sources.sort((left, right) => left.label.localeCompare(right.label));
1975
- const score = sources[0].value;
1976
- if (sources.some((source) => source.value !== score)) {
1977
- const details = sources.map((source) => `${source.label}=${source.value}`).join(", ");
1978
- throw new ValidationError(
1979
- `fromOtelSpans: conflicting task-quality scores for run '${runId}': ${details}`
1980
- );
1981
- }
1982
- return score;
1819
+ const orderedSpans = orderSpans(spans);
1820
+ const sources = [];
1821
+ if (scoreForRun) {
1822
+ const supplied = scoreForRun(runId, orderedSpans);
1823
+ if (supplied !== void 0) {
1824
+ if (typeof supplied !== "number" || !Number.isFinite(supplied)) throw new ValidationError(`fromOtelSpans: scoreForRun returned a non-finite number for run '${runId}'`);
1825
+ sources.push({
1826
+ label: "scoreForRun",
1827
+ value: supplied
1828
+ });
1829
+ }
1830
+ }
1831
+ for (const span of orderedSpans) {
1832
+ const role = errorRoleForSpan(span);
1833
+ if (span.parentSpanId && role !== "EVALUATOR") continue;
1834
+ if (role === "EVALUATOR" && span.status?.code === "ERROR") continue;
1835
+ for (const key of TASK_SCORE_ATTR_KEYS) {
1836
+ if (!Object.hasOwn(span.attributes, key)) continue;
1837
+ sources.push({
1838
+ label: `span '${span.spanId}' attribute '${key}'`,
1839
+ value: parseTaskScoreAttribute(runId, span.spanId, key, span.attributes[key])
1840
+ });
1841
+ }
1842
+ }
1843
+ if (sources.length === 0) return void 0;
1844
+ sources.sort((left, right) => left.label.localeCompare(right.label));
1845
+ const score = sources[0].value;
1846
+ if (sources.some((source) => source.value !== score)) throw new ValidationError(`fromOtelSpans: conflicting task-quality scores for run '${runId}': ${sources.map((source) => `${source.label}=${source.value}`).join(", ")}`);
1847
+ return score;
1983
1848
  }
1984
1849
  function parseTaskScoreAttribute(runId, spanId, key, value) {
1985
- const source = `span '${spanId}' attribute '${key}'`;
1986
- if (typeof value === "string") {
1987
- if (value.trim().length === 0) {
1988
- throw new ValidationError(
1989
- `fromOtelSpans: ${source} is blank for run '${runId}'; task quality must be finite`
1990
- );
1991
- }
1992
- const parsed = Number(value);
1993
- if (Number.isFinite(parsed)) return parsed;
1994
- } else if (typeof value === "number" && Number.isFinite(value)) {
1995
- return value;
1996
- }
1997
- throw new ValidationError(
1998
- `fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`
1999
- );
1850
+ const source = `span '${spanId}' attribute '${key}'`;
1851
+ if (typeof value === "string") {
1852
+ if (value.trim().length === 0) throw new ValidationError(`fromOtelSpans: ${source} is blank for run '${runId}'; task quality must be finite`);
1853
+ const parsed = Number(value);
1854
+ if (Number.isFinite(parsed)) return parsed;
1855
+ } else if (typeof value === "number" && Number.isFinite(value)) return value;
1856
+ throw new ValidationError(`fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`);
2000
1857
  }
2001
1858
  function orderSpans(spans) {
2002
- return [...spans].sort(
2003
- (left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId)
2004
- );
1859
+ return [...spans].sort((left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId));
2005
1860
  }
2006
1861
  function parseUnixNano(value) {
2007
- try {
2008
- return BigInt(value);
2009
- } catch {
2010
- throw new ValidationError(`fromOtelSpans: invalid Unix nanosecond timestamp '${value}'`);
2011
- }
1862
+ try {
1863
+ return BigInt(value);
1864
+ } catch {
1865
+ throw new ValidationError(`fromOtelSpans: invalid Unix nanosecond timestamp '${value}'`);
1866
+ }
2012
1867
  }
2013
1868
  function compareUnixNano(left, right) {
2014
- const leftValue = parseUnixNano(left);
2015
- const rightValue = parseUnixNano(right);
2016
- return leftValue < rightValue ? -1 : leftValue > rightValue ? 1 : 0;
1869
+ const leftValue = parseUnixNano(left);
1870
+ const rightValue = parseUnixNano(right);
1871
+ return leftValue < rightValue ? -1 : leftValue > rightValue ? 1 : 0;
2017
1872
  }
2018
1873
  function unixNanoDurationMs(start, end) {
2019
- const delta = parseUnixNano(end) - parseUnixNano(start);
2020
- if (delta <= 0n) return 0;
2021
- const wholeMs = delta / 1000000n;
2022
- const fractionalMs = delta % 1000000n;
2023
- const value = Number(wholeMs) + Number(fractionalMs) / 1e6;
2024
- if (!Number.isSafeInteger(Number(wholeMs))) {
2025
- throw new ValidationError("fromOtelSpans: span duration exceeds the safe millisecond range");
2026
- }
2027
- return value;
1874
+ const delta = parseUnixNano(end) - parseUnixNano(start);
1875
+ if (delta <= 0n) return 0;
1876
+ const wholeMs = delta / 1000000n;
1877
+ const fractionalMs = delta % 1000000n;
1878
+ const value = Number(wholeMs) + Number(fractionalMs) / 1e6;
1879
+ if (!Number.isSafeInteger(Number(wholeMs))) throw new ValidationError("fromOtelSpans: span duration exceeds the safe millisecond range");
1880
+ return value;
2028
1881
  }
2029
1882
  function collectNumericAttrs(spans) {
2030
- const raw = {};
2031
- for (const span of spans) {
2032
- for (const [k, v] of Object.entries(span.attributes)) {
2033
- if (typeof v === "number" && Number.isFinite(v)) raw[k] = v;
2034
- }
2035
- }
2036
- return raw;
2037
- }
2038
- export {
2039
- FileSystemOutcomeStore,
2040
- InMemoryOutcomeStore,
2041
- REFERENCE_EQUIVALENCE_INPUT_LIMITS,
2042
- REFERENCE_EQUIVALENCE_JUDGE_VERSION,
2043
- SelfImproveRunError,
2044
- analyzeRuns,
2045
- buildDefaultAnalystRegistry,
2046
- buildEvidenceVector,
2047
- campaignSplitDigest,
2048
- compareOptimizationMethods,
2049
- composeGate,
2050
- createChatClient,
2051
- createReferenceEquivalenceJudge,
2052
- defaultProductionGate,
2053
- defineAgentEval,
2054
- diffGenerations,
2055
- diffRunBaselineToWinner,
2056
- diffRuns,
2057
- evalReportingSuite,
2058
- evaluatePairedMeasurements,
2059
- externalTextOptimizationMethod,
2060
- fromClaudeCodeSession,
2061
- fromCodexSession,
2062
- fromFeedbackTable,
2063
- fromKimiCodeSession,
2064
- fromOpenCodeSession,
2065
- fromOtelSpans,
2066
- fromPiSession,
2067
- fromPigraphSession,
2068
- fromRunRecordDir,
2069
- fsCampaignStorage,
2070
- gepaOptimizationMethod,
2071
- heldOutGate,
2072
- inMemoryCampaignStorage,
2073
- llmJudge,
2074
- measuredComparisonFromCandidateExperiment,
2075
- observeCodeAgentSession,
2076
- paretoPolicy,
2077
- paretoSignificanceGate,
2078
- parseAgentTrace,
2079
- parseCodeAgentJsonl,
2080
- partitionRunsByAuthoringModel,
2081
- runCampaign,
2082
- runCandidateExperiment,
2083
- runEval,
2084
- runImprovementLoop,
2085
- runReferenceEquivalenceJudge,
2086
- sealCandidateBenchmarkSuite,
2087
- sealCandidateBenchmarkTask,
2088
- sealCandidateExperiment,
2089
- selfImprove,
2090
- skillOptOptimizationMethod,
2091
- summarizeExecution,
2092
- verifyCandidateBenchmarkSuite,
2093
- verifyCandidateBenchmarkSuiteInputs,
2094
- verifyCandidateBenchmarkTask,
2095
- verifyCandidateExperiment,
2096
- verifyCandidateExperimentComparison
2097
- };
1883
+ const raw = {};
1884
+ for (const span of spans) for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "number" && Number.isFinite(v)) raw[k] = v;
1885
+ return raw;
1886
+ }
1887
+ //#endregion
1888
+ export { FileSystemOutcomeStore, InMemoryOutcomeStore, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, SelfImproveRunError, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
1889
+
2098
1890
  //# sourceMappingURL=index.js.map