@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -1,2097 +1,1890 @@
1
- import {
2
- fromClaudeCodeSession,
3
- fromCodexSession,
4
- fromKimiCodeSession,
5
- fromOpenCodeSession,
6
- fromPiSession,
7
- fromPigraphSession,
8
- observeCodeAgentSession,
9
- parseCodeAgentJsonl
10
- } from "../chunk-SZLVEKMJ.js";
11
- import {
12
- createHostedClient
13
- } from "../chunk-DRYIUNWY.js";
14
- import {
15
- analyzeRuns,
16
- summarizeExecution
17
- } from "../chunk-NACAGYSY.js";
18
- import {
19
- REFERENCE_EQUIVALENCE_INPUT_LIMITS,
20
- REFERENCE_EQUIVALENCE_JUDGE_VERSION,
21
- assertOptimizationResult,
22
- buildEvidenceVector,
23
- compareOptimizationMethods,
24
- composeGate,
25
- createReferenceEquivalenceJudge,
26
- defaultProductionGate,
27
- emitLoopProvenance,
28
- externalTextOptimizationMethod,
29
- gepaOptimizationMethod,
30
- heldOutGate,
31
- heldoutSignificance,
32
- llmJudge,
33
- loopProvenanceArgsFromResult,
34
- paretoPolicy,
35
- paretoSignificanceGate,
36
- powerPreflight,
37
- runEval,
38
- runImprovementLoop,
39
- runReferenceEquivalenceJudge,
40
- skillOptOptimizationMethod,
41
- surfaceContentHash,
42
- surfaceHash
43
- } from "../chunk-NKAGIDE2.js";
44
- import {
45
- campaignSplitDigest,
46
- createRunCostLedger,
47
- fsCampaignStorage,
48
- inMemoryCampaignStorage,
49
- resolveRunDir,
50
- runCampaign
51
- } from "../chunk-EZJEIH2R.js";
52
- import {
53
- buildDefaultAnalystRegistry,
54
- createChatClient
55
- } from "../chunk-DJKY2TSY.js";
56
- import "../chunk-HHWE3POT.js";
57
- import "../chunk-WGXIEX7P.js";
58
- import {
59
- FileSystemOutcomeStore,
60
- InMemoryOutcomeStore
61
- } from "../chunk-3RF76KTD.js";
62
- import "../chunk-NYLOYM6N.js";
63
- import {
64
- campaignCellExecutionEvidence,
65
- campaignCellJudgeDimensions,
66
- campaignCellTaskScore,
67
- campaignCellToRunRecord
68
- } from "../chunk-2MKQIFS4.js";
69
- import "../chunk-PBE2LOSS.js";
70
- import "../chunk-VLOATJQ2.js";
71
- import "../chunk-DPUHNQLN.js";
72
- import {
73
- pairedBootstrap
74
- } from "../chunk-MHELPNRP.js";
75
- import "../chunk-WS3NZZQQ.js";
76
- import "../chunk-VI2UW6B6.js";
77
- import {
78
- readTaskFailureLabels,
79
- recordAggregateMeasurements,
80
- summarizeExecutionMeasurements,
81
- summarizeTraceErrors
82
- } from "../chunk-7ZZMD7UK.js";
83
- import "../chunk-PXE2VKMX.js";
84
- import "../chunk-ZET2UAYW.js";
85
- import "../chunk-GGE4NNQT.js";
86
- import {
87
- classifyOtlpSpanRole,
88
- isOtlpModelCall
89
- } from "../chunk-P6FYH6K4.js";
90
- import "../chunk-PC4UYEBM.js";
91
- import {
92
- modelHasSnapshot,
93
- parseRunRecordSafe
94
- } from "../chunk-2JX3CFMB.js";
95
- import "../chunk-MA6HLL3S.js";
96
- import {
97
- ValidationError
98
- } from "../chunk-ONWEPEDO.js";
99
- import {
100
- LLM_MODEL_ATTR_KEYS,
101
- SPAN_KIND_ATTR_KEYS
102
- } from "../chunk-K4DBDHLK.js";
103
- import "../chunk-PZ5AY32C.js";
104
-
105
- // src/contract/self-improve.ts
1
+ import { s as ValidationError } from "../errors-8YnH8WlF.js";
2
+ import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-BuoE80Dq.js";
3
+ import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-C-vFCSEc.js";
4
+ import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
5
+ import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
6
+ import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-D4ODwFVV.js";
7
+ import { v as pairedBootstrap } from "../statistics-CnnxdpOg.js";
8
+ import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-CQZlB3et.js";
9
+ import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-C1CavBMk.js";
10
+ import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-qipEpKvY.js";
11
+ import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
12
+ import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSession, n as fromCodexSession, o as fromPigraphSession, r as fromKimiCodeSession, s as parseCodeAgentJsonl, t as fromClaudeCodeSession } from "../code-agent-session-BjkMTQ7H.js";
13
+ import { t as createHostedClient } from "../client-CYzbdJOZ.js";
14
+ import { dirname, join } from "node:path";
15
+ import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
16
+ import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, omitTopLevelDigest } from "@tangle-network/agent-interface";
17
+ //#region src/contract/self-improve.ts
18
+ /**
19
+ * Run one complete improvement job.
20
+ *
21
+ * A caller-owned `proposer` can generate candidates across local generations.
22
+ * An external `method`, such as official GEPA or SkillOpt, owns its complete
23
+ * search and returns one candidate. Both paths remeasure the selected candidate
24
+ * against cases that candidate generation never receives.
25
+ */
26
+ /** Failed self-improvement run with an immutable receipt snapshot. */
106
27
  var SelfImproveRunError = class extends Error {
107
- cost;
108
- receipts;
109
- constructor(cause, ledger) {
110
- const original = cause instanceof Error ? cause : new Error(String(cause));
111
- super(original.message, { cause: original });
112
- this.name = "SelfImproveRunError";
113
- this.cost = ledger.summary();
114
- this.receipts = ledger.list();
115
- }
28
+ cost;
29
+ receipts;
30
+ constructor(cause, ledger) {
31
+ const original = cause instanceof Error ? cause : new Error(String(cause));
32
+ super(original.message, { cause: original });
33
+ this.name = "SelfImproveRunError";
34
+ this.cost = ledger.summary();
35
+ this.receipts = ledger.list();
36
+ }
116
37
  };
117
38
  function assertSelfImproveSearchMode(opts) {
118
- if (opts.method && opts.proposer) {
119
- throw new Error("selfImprove: method and proposer are mutually exclusive");
120
- }
121
- if (!opts.method) {
122
- if (opts.selectionScenarios !== void 0) {
123
- throw new Error("selfImprove: selectionScenarios requires method");
124
- }
125
- return;
126
- }
127
- if (typeof opts.method.name !== "string" || !opts.method.name.trim() || opts.method.name.trim() !== opts.method.name || typeof opts.method.optimize !== "function") {
128
- throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
129
- }
130
- const budget = opts.budget;
131
- if (budget?.generations !== void 0 && budget.generations !== 1) {
132
- throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
133
- }
134
- if (budget?.populationSize !== void 0 && budget.populationSize !== 1) {
135
- throw new Error(
136
- "selfImprove: method owns its candidates; budget.populationSize must be 1 when set"
137
- );
138
- }
139
- if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) {
140
- throw new Error(
141
- "selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode"
142
- );
143
- }
39
+ if (opts.method && opts.proposer) throw new Error("selfImprove: method and proposer are mutually exclusive");
40
+ if (!opts.method) {
41
+ if (opts.selectionScenarios !== void 0) throw new Error("selfImprove: selectionScenarios requires method");
42
+ return;
43
+ }
44
+ if (typeof opts.method.name !== "string" || !opts.method.name.trim() || opts.method.name.trim() !== opts.method.name || typeof opts.method.optimize !== "function") throw new Error("selfImprove: method must have a trimmed name and optimize(input)");
45
+ const budget = opts.budget;
46
+ if (budget?.generations !== void 0 && budget.generations !== 1) throw new Error("selfImprove: method owns its rounds; budget.generations must be 1 when set");
47
+ if (budget?.populationSize !== void 0 && budget.populationSize !== 1) throw new Error("selfImprove: method owns its candidates; budget.populationSize must be 1 when set");
48
+ if (budget?.candidateConcurrency !== void 0 || budget?.maxImprovementShots !== void 0 || opts.analyzeGeneration !== void 0 || opts.findings !== void 0) throw new Error("selfImprove: candidateConcurrency, maxImprovementShots, analyzeGeneration, and findings apply only to proposer mode");
144
49
  }
145
50
  function splitMethodPartitions(searchScenarios, explicitSelection, fraction) {
146
- if (!Number.isFinite(fraction) || fraction <= 0 || fraction >= 1) {
147
- throw new Error("selfImprove: budget.selectionFraction must be in (0, 1)");
148
- }
149
- const byId = /* @__PURE__ */ new Map();
150
- for (const scenario of searchScenarios) {
151
- if (byId.has(scenario.id)) {
152
- throw new Error(`selfImprove: duplicate scenario id '${scenario.id}'`);
153
- }
154
- byId.set(scenario.id, scenario);
155
- }
156
- if (explicitSelection) {
157
- if (explicitSelection.length === 0) {
158
- throw new Error("selfImprove: selectionScenarios must not be empty");
159
- }
160
- const selectionIds = /* @__PURE__ */ new Set();
161
- for (const scenario of explicitSelection) {
162
- if (!byId.has(scenario.id)) {
163
- throw new Error(
164
- `selfImprove: selection scenario '${scenario.id}' is absent from the non-final cases`
165
- );
166
- }
167
- if (selectionIds.has(scenario.id)) {
168
- throw new Error(`selfImprove: duplicate selection scenario id '${scenario.id}'`);
169
- }
170
- selectionIds.add(scenario.id);
171
- }
172
- const train = searchScenarios.filter((scenario) => !selectionIds.has(scenario.id));
173
- if (train.length === 0) {
174
- throw new Error("selfImprove: method train split is empty");
175
- }
176
- return {
177
- train,
178
- selection: explicitSelection.map((scenario) => byId.get(scenario.id))
179
- };
180
- }
181
- if (searchScenarios.length < 2) {
182
- throw new Error("selfImprove: method requires at least two non-final scenarios");
183
- }
184
- const sorted = [...searchScenarios].sort(
185
- (a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id)
186
- );
187
- const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
188
- return {
189
- selection: sorted.slice(0, count),
190
- train: sorted.slice(count)
191
- };
51
+ if (!Number.isFinite(fraction) || fraction <= 0 || fraction >= 1) throw new Error("selfImprove: budget.selectionFraction must be in (0, 1)");
52
+ const byId = /* @__PURE__ */ new Map();
53
+ for (const scenario of searchScenarios) {
54
+ if (byId.has(scenario.id)) throw new Error(`selfImprove: duplicate scenario id '${scenario.id}'`);
55
+ byId.set(scenario.id, scenario);
56
+ }
57
+ if (explicitSelection) {
58
+ if (explicitSelection.length === 0) throw new Error("selfImprove: selectionScenarios must not be empty");
59
+ const selectionIds = /* @__PURE__ */ new Set();
60
+ for (const scenario of explicitSelection) {
61
+ if (!byId.has(scenario.id)) throw new Error(`selfImprove: selection scenario '${scenario.id}' is absent from the non-final cases`);
62
+ if (selectionIds.has(scenario.id)) throw new Error(`selfImprove: duplicate selection scenario id '${scenario.id}'`);
63
+ selectionIds.add(scenario.id);
64
+ }
65
+ const train = searchScenarios.filter((scenario) => !selectionIds.has(scenario.id));
66
+ if (train.length === 0) throw new Error("selfImprove: method train split is empty");
67
+ return {
68
+ train,
69
+ selection: explicitSelection.map((scenario) => byId.get(scenario.id))
70
+ };
71
+ }
72
+ if (searchScenarios.length < 2) throw new Error("selfImprove: method requires at least two non-final scenarios");
73
+ const sorted = [...searchScenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
74
+ const count = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
75
+ return {
76
+ selection: sorted.slice(0, count),
77
+ train: sorted.slice(count)
78
+ };
192
79
  }
193
80
  function safeRunComponent(value) {
194
- return value.replace(/[^a-zA-Z0-9._-]/g, "_");
81
+ return value.replace(/[^a-zA-Z0-9._-]/g, "_");
195
82
  }
196
83
  function stableScenarioHash(value) {
197
- let hash = 2166136261 >>> 0;
198
- for (let index = 0; index < value.length; index++) {
199
- hash ^= value.charCodeAt(index);
200
- hash = Math.imul(hash, 16777619) >>> 0;
201
- }
202
- return hash;
203
- }
84
+ let hash = 2166136261;
85
+ for (let index = 0; index < value.length; index++) {
86
+ hash ^= value.charCodeAt(index);
87
+ hash = Math.imul(hash, 16777619) >>> 0;
88
+ }
89
+ return hash;
90
+ }
91
+ /**
92
+ * Deterministic train/holdout split by a stable hash of `scenario.id`,
93
+ * so the same scenario set always splits the same way across runs.
94
+ */
204
95
  function splitTrainHoldout(scenarios, fraction) {
205
- const sorted = [...scenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
206
- const nHoldout = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
207
- return {
208
- holdout: sorted.slice(0, nHoldout),
209
- train: sorted.slice(nHoldout)
210
- };
96
+ const sorted = [...scenarios].sort((a, b) => stableScenarioHash(a.id) - stableScenarioHash(b.id));
97
+ const nHoldout = Math.max(1, Math.min(sorted.length - 1, Math.round(sorted.length * fraction)));
98
+ return {
99
+ holdout: sorted.slice(0, nHoldout),
100
+ train: sorted.slice(nHoldout)
101
+ };
211
102
  }
212
103
  function meanComposite(byScenario) {
213
- const perScenario = {};
214
- const values = [];
215
- for (const [id, agg] of Object.entries(byScenario)) {
216
- perScenario[id] = agg.meanComposite;
217
- values.push(agg.meanComposite);
218
- }
219
- return {
220
- compositeMean: values.length === 0 ? 0 : values.reduce((s, v) => s + v, 0) / values.length,
221
- perScenario
222
- };
223
- }
104
+ const perScenario = {};
105
+ const values = [];
106
+ for (const [id, agg] of Object.entries(byScenario)) {
107
+ perScenario[id] = agg.meanComposite;
108
+ values.push(agg.meanComposite);
109
+ }
110
+ return {
111
+ compositeMean: values.length === 0 ? 0 : values.reduce((s, v) => s + v, 0) / values.length,
112
+ perScenario
113
+ };
114
+ }
115
+ /**
116
+ * Latest search campaign measured for the winner surface; the baseline search
117
+ * campaign when the winner IS the baseline. Used by the deferred-holdout
118
+ * summary, where no holdout campaign exists to summarize.
119
+ */
224
120
  function winnerSearchCampaign(result) {
225
- for (let i = result.generations.length - 1; i >= 0; i--) {
226
- const measured = result.generations[i]?.surfaces.find(
227
- (s) => s.surfaceHash === result.winnerSurfaceHash
228
- );
229
- if (measured) return measured.campaign;
230
- }
231
- return result.baselineCampaign;
232
- }
121
+ for (let i = result.generations.length - 1; i >= 0; i--) {
122
+ const measured = result.generations[i]?.surfaces.find((s) => s.surfaceHash === result.winnerSurfaceHash);
123
+ if (measured) return measured.campaign;
124
+ }
125
+ return result.baselineCampaign;
126
+ }
127
+ /**
128
+ * One-shot self-improvement loop. See module docstring for defaults +
129
+ * extension points.
130
+ *
131
+ * @example Minimum:
132
+ *
133
+ * const result = await selfImprove({
134
+ * agent: (surface, scenario, ctx) => myAgent(surface, scenario, ctx.signal),
135
+ * scenarios,
136
+ * judge,
137
+ * baselineSurface: DEFAULT_PROMPT,
138
+ * proposer,
139
+ * })
140
+ * console.log(`lift: ${result.lift.toFixed(3)} (${result.gateDecision})`)
141
+ *
142
+ * @example Distributed (workers in three regions):
143
+ *
144
+ * await selfImprove({
145
+ * agent: httpDispatch({ resolveUrl: ({ placement }) => REGION_URLS[placement!] }),
146
+ * scenarios,
147
+ * judge,
148
+ * baselineSurface: DEFAULT_PROMPT,
149
+ * cellPlacement: ({ scenario }) => scenario.region,
150
+ * budget: { maxConcurrency: 12 },
151
+ * })
152
+ */
233
153
  async function selfImprove(opts) {
234
- const startedAt = Date.now();
235
- const requestedRunDir = opts.runDir ?? (opts.method ? `.agent-eval/runs/self-improve-${startedAt}` : `mem://selfImprove-${startedAt}`);
236
- const runDir = resolveRunDir(requestedRunDir);
237
- const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
238
- const costLedger = createRunCostLedger({
239
- storage,
240
- runDir,
241
- costCeilingUsd: opts.budget?.dollars
242
- });
243
- try {
244
- return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
245
- } catch (error) {
246
- throw new SelfImproveRunError(error, costLedger);
247
- }
154
+ const startedAt = Date.now();
155
+ const runDir = resolveRunDir(opts.runDir ?? (opts.method ? `.agent-eval/runs/self-improve-${startedAt}` : `mem://selfImprove-${startedAt}`));
156
+ const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
157
+ const costLedger = createRunCostLedger({
158
+ storage,
159
+ runDir,
160
+ costCeilingUsd: opts.budget?.dollars
161
+ });
162
+ try {
163
+ return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
164
+ } catch (error) {
165
+ throw new SelfImproveRunError(error, costLedger);
166
+ }
248
167
  }
249
168
  async function runSelfImprove(opts, costLedger, startedAt, runDir, storage) {
250
- const budget = opts.budget ?? {};
251
- assertSelfImproveSearchMode(opts);
252
- const generations = opts.method ? 1 : budget.generations ?? 3;
253
- const populationSize = opts.method ? 1 : budget.populationSize ?? 2;
254
- const maxConcurrency = budget.maxConcurrency ?? 2;
255
- const holdoutFraction = budget.holdoutFraction ?? 0.25;
256
- const holdoutMode = budget.holdout ?? "measured";
257
- const holdoutDeferred = holdoutMode === "deferred";
258
- const expectUsage = opts.expectUsage ?? "assert";
259
- const explicitHoldout = budget.holdoutScenarios;
260
- const { train, holdout } = explicitHoldout ? {
261
- train: opts.scenarios.filter((s) => !explicitHoldout.some((h) => h.id === s.id)),
262
- holdout: explicitHoldout
263
- } : holdoutDeferred ? { train: opts.scenarios, holdout: [] } : splitTrainHoldout(opts.scenarios, holdoutFraction);
264
- if (train.length === 0) {
265
- throw new Error(
266
- "selfImprove: train split is empty. Reduce holdoutFraction or pass more scenarios."
267
- );
268
- }
269
- if (holdout.length === 0 && !holdoutDeferred) {
270
- throw new Error("selfImprove: holdout split is empty. Pass more scenarios.");
271
- }
272
- if (generations > 0 && !opts.proposer && !opts.method) {
273
- throw new Error(
274
- "selfImprove: method or proposer is required when budget.generations is greater than zero"
275
- );
276
- }
277
- let optimizationResult;
278
- const methodPartitions = opts.method ? splitMethodPartitions(train, opts.selectionScenarios, budget.selectionFraction ?? 0.25) : void 0;
279
- const proposer = opts.method ? {
280
- kind: `method:${opts.method.name}`,
281
- propose: async (context) => {
282
- if (context.generation > 0) return [];
283
- const result2 = await opts.method.optimize(
284
- Object.freeze({
285
- baselineSurface: structuredClone(context.currentSurface),
286
- trainScenarios: Object.freeze(
287
- methodPartitions.train.map((scenario) => structuredClone(scenario))
288
- ),
289
- selectionScenarios: Object.freeze(
290
- methodPartitions.selection.map((scenario) => structuredClone(scenario))
291
- ),
292
- dispatchWithSurface: opts.agent,
293
- judges: Object.freeze([opts.judge]),
294
- runDir: `${runDir}/optimization/${safeRunComponent(opts.method.name)}`,
295
- seed: 42,
296
- runOptions: Object.freeze({
297
- storage,
298
- maxConcurrency,
299
- reps: budget.reps,
300
- dispatchTimeoutMs: opts.dispatchTimeoutMs,
301
- expectUsage,
302
- costCeiling: budget.dollars
303
- }),
304
- costLedger
305
- })
306
- );
307
- assertOptimizationResult(opts.method.name, result2);
308
- optimizationResult = structuredClone(result2);
309
- return [
310
- {
311
- surface: structuredClone(result2.winnerSurface),
312
- label: opts.method.name,
313
- rationale: `${opts.method.name} selected this surface without final cases.`
314
- }
315
- ];
316
- }
317
- } : opts.proposer ?? {
318
- kind: "baseline-only",
319
- propose: async () => []
320
- };
321
- const gate = opts.gate ?? defaultProductionGate({
322
- holdoutScenarios: holdout,
323
- deltaThreshold: 0.05
324
- });
325
- if (opts.onProgress) {
326
- opts.onProgress({ kind: "baseline.started", scenarios: opts.scenarios.length });
327
- }
328
- const result = await runImprovementLoop({
329
- scenarios: train,
330
- baselineSurface: opts.baselineSurface,
331
- premeasuredBaseline: opts.premeasuredBaseline,
332
- dispatchWithSurface: opts.agent,
333
- proposer,
334
- judges: [opts.judge],
335
- populationSize,
336
- maxGenerations: generations,
337
- candidateConcurrency: budget.candidateConcurrency,
338
- reps: budget.reps,
339
- maxImprovementShots: budget.maxImprovementShots,
340
- holdoutScenarios: holdout,
341
- holdout: holdoutMode,
342
- gate,
343
- neutralize: opts.neutralize,
344
- autoOnPromote: opts.autoOnPromote ?? "none",
345
- ghOwner: opts.ghOwner,
346
- ghRepo: opts.ghRepo,
347
- storage,
348
- runDir,
349
- maxConcurrency,
350
- cellPlacement: opts.cellPlacement,
351
- dispatchTimeoutMs: opts.dispatchTimeoutMs,
352
- costLedger,
353
- expectUsage,
354
- labeledStore: opts.labeledStore,
355
- captureSource: opts.captureSource,
356
- analyzeGeneration: opts.analyzeGeneration,
357
- findings: opts.findings,
358
- selectionRankKey: opts.selectionRankKey
359
- });
360
- const reportSplit = holdoutDeferred ? "search" : "holdout";
361
- const reportBaselineCampaign = holdoutDeferred ? result.baselineCampaign : result.baselineOnHoldout;
362
- const reportWinnerCampaign = holdoutDeferred ? winnerSearchCampaign(result) : result.winnerOnHoldout;
363
- const baseline = meanComposite(reportBaselineCampaign.aggregates.byScenario);
364
- const winnerStats = meanComposite(reportWinnerCampaign.aggregates.byScenario);
365
- let power;
366
- const baselineHoldoutComposites = result.baselineOnHoldout.cells.filter((cell) => !cell.error).map((cell) => {
367
- const scores = Object.values(cell.judgeScores);
368
- return scores.length === 0 ? Number.NaN : scores.reduce((sum, s) => sum + s.composite, 0) / scores.length;
369
- }).filter((v) => Number.isFinite(v));
370
- if (baselineHoldoutComposites.length >= 3) {
371
- power = powerPreflight({
372
- baselineComposites: baselineHoldoutComposites,
373
- sharedScorerChannel: true
374
- });
375
- if (opts.onProgress) {
376
- opts.onProgress({
377
- kind: "power.estimated",
378
- n: power.n,
379
- sd: power.sd,
380
- mde: power.mde,
381
- underpowered: power.underpowered
382
- });
383
- }
384
- if (power.underpowered && generations > 0) {
385
- console.warn(`[selfImprove] ${power.recommendation}`);
386
- }
387
- }
388
- if (opts.onProgress) {
389
- opts.onProgress({
390
- kind: "baseline.completed",
391
- compositeMean: baseline.compositeMean,
392
- durationMs: Date.now() - startedAt
393
- });
394
- opts.onProgress({
395
- kind: "gate.decided",
396
- decision: result.gateResult.decision,
397
- // Deferred holdout has no held-out measurement: in that mode the summary
398
- // stats are search-split numbers, and emitting their delta as `lift`
399
- // would misreport a train-split delta as a held-out one. Omit instead.
400
- ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean }
401
- });
402
- }
403
- const cost = result.cost;
404
- const totalCost = cost.totalCostUsd;
405
- const insight = await analyzeRuns({
406
- runs: [
407
- ...cellsToRunRecords(
408
- reportBaselineCampaign.cells,
409
- "baseline",
410
- runDir,
411
- opts.baselineSurface,
412
- reportSplit,
413
- opts.model
414
- ),
415
- ...cellsToRunRecords(
416
- reportWinnerCampaign.cells,
417
- "winner",
418
- runDir,
419
- result.winnerSurface,
420
- reportSplit,
421
- opts.model
422
- )
423
- ],
424
- baselineCandidateId: "baseline",
425
- candidateCandidateId: "winner"
426
- });
427
- const durationMs = Date.now() - startedAt;
428
- const { record: provenance } = await emitLoopProvenance({
429
- ...loopProvenanceArgsFromResult({
430
- runId: `${runDir}#${startedAt}`,
431
- runDir,
432
- timestamp: new Date(startedAt).toISOString(),
433
- baselineSurface: opts.baselineSurface,
434
- result,
435
- costReceipts: costLedger.list(),
436
- totalCostUsd: totalCost,
437
- totalDurationMs: durationMs
438
- }),
439
- ...optimizationResult ? {
440
- optimizationMethod: {
441
- name: opts.method.name,
442
- cost: structuredClone(optimizationResult.cost),
443
- ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
444
- ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
445
- }
446
- } : {},
447
- storage,
448
- hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
449
- });
450
- if (opts.onProvenance) opts.onProvenance(provenance);
451
- const summary = {
452
- baseline,
453
- winner: {
454
- ...winnerStats,
455
- surface: result.winnerSurface,
456
- ...result.winnerLabel ? { label: result.winnerLabel } : {},
457
- ...result.winnerRationale ? { rationale: result.winnerRationale } : {}
458
- },
459
- ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
460
- diff: result.promotedDiff,
461
- provenance,
462
- gateDecision: result.gateResult.decision,
463
- generationsExplored: result.generations.length,
464
- durationMs,
465
- totalCostUsd: totalCost,
466
- cost,
467
- receipts: costLedger.list(),
468
- ...optimizationResult ? {
469
- optimization: {
470
- name: opts.method.name,
471
- cost: structuredClone(optimizationResult.cost),
472
- ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
473
- ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
474
- }
475
- } : {},
476
- insight,
477
- ...power ? { power } : {},
478
- raw: result
479
- };
480
- if (opts.hostedTenant) {
481
- try {
482
- await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
483
- } catch (err) {
484
- const msg = err instanceof Error ? err.message : String(err);
485
- console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
486
- }
487
- }
488
- return summary;
169
+ const budget = opts.budget ?? {};
170
+ assertSelfImproveSearchMode(opts);
171
+ const generations = opts.method ? 1 : budget.generations ?? 3;
172
+ const populationSize = opts.method ? 1 : budget.populationSize ?? 2;
173
+ const maxConcurrency = budget.maxConcurrency ?? 2;
174
+ const holdoutFraction = budget.holdoutFraction ?? .25;
175
+ const holdoutMode = budget.holdout ?? "measured";
176
+ const holdoutDeferred = holdoutMode === "deferred";
177
+ const expectUsage = opts.expectUsage ?? "assert";
178
+ const explicitHoldout = budget.holdoutScenarios;
179
+ const { train, holdout } = explicitHoldout ? {
180
+ train: opts.scenarios.filter((s) => !explicitHoldout.some((h) => h.id === s.id)),
181
+ holdout: explicitHoldout
182
+ } : holdoutDeferred ? {
183
+ train: opts.scenarios,
184
+ holdout: []
185
+ } : splitTrainHoldout(opts.scenarios, holdoutFraction);
186
+ if (train.length === 0) throw new Error("selfImprove: train split is empty. Reduce holdoutFraction or pass more scenarios.");
187
+ if (holdout.length === 0 && !holdoutDeferred) throw new Error("selfImprove: holdout split is empty. Pass more scenarios.");
188
+ if (generations > 0 && !opts.proposer && !opts.method) throw new Error("selfImprove: method or proposer is required when budget.generations is greater than zero");
189
+ let optimizationResult;
190
+ const methodPartitions = opts.method ? splitMethodPartitions(train, opts.selectionScenarios, budget.selectionFraction ?? .25) : void 0;
191
+ const proposer = opts.method ? {
192
+ kind: `method:${opts.method.name}`,
193
+ propose: async (context) => {
194
+ if (context.generation > 0) return [];
195
+ const result = await opts.method.optimize(Object.freeze({
196
+ baselineSurface: structuredClone(context.currentSurface),
197
+ trainScenarios: Object.freeze(methodPartitions.train.map((scenario) => structuredClone(scenario))),
198
+ selectionScenarios: Object.freeze(methodPartitions.selection.map((scenario) => structuredClone(scenario))),
199
+ dispatchWithSurface: opts.agent,
200
+ judges: Object.freeze([opts.judge]),
201
+ runDir: `${runDir}/optimization/${safeRunComponent(opts.method.name)}`,
202
+ seed: 42,
203
+ runOptions: Object.freeze({
204
+ storage,
205
+ maxConcurrency,
206
+ reps: budget.reps,
207
+ dispatchTimeoutMs: opts.dispatchTimeoutMs,
208
+ expectUsage,
209
+ costCeiling: budget.dollars
210
+ }),
211
+ costLedger
212
+ }));
213
+ assertOptimizationResult(opts.method.name, result);
214
+ optimizationResult = structuredClone(result);
215
+ return [{
216
+ surface: structuredClone(result.winnerSurface),
217
+ label: opts.method.name,
218
+ rationale: `${opts.method.name} selected this surface without final cases.`
219
+ }];
220
+ }
221
+ } : opts.proposer ?? {
222
+ kind: "baseline-only",
223
+ propose: async () => []
224
+ };
225
+ const gate = opts.gate ?? defaultProductionGate({
226
+ holdoutScenarios: holdout,
227
+ deltaThreshold: .05
228
+ });
229
+ if (opts.onProgress) opts.onProgress({
230
+ kind: "baseline.started",
231
+ scenarios: opts.scenarios.length
232
+ });
233
+ const result = await runImprovementLoop({
234
+ scenarios: train,
235
+ baselineSurface: opts.baselineSurface,
236
+ premeasuredBaseline: opts.premeasuredBaseline,
237
+ dispatchWithSurface: opts.agent,
238
+ proposer,
239
+ judges: [opts.judge],
240
+ populationSize,
241
+ maxGenerations: generations,
242
+ candidateConcurrency: budget.candidateConcurrency,
243
+ reps: budget.reps,
244
+ maxImprovementShots: budget.maxImprovementShots,
245
+ holdoutScenarios: holdout,
246
+ holdout: holdoutMode,
247
+ gate,
248
+ neutralize: opts.neutralize,
249
+ autoOnPromote: opts.autoOnPromote ?? "none",
250
+ ghOwner: opts.ghOwner,
251
+ ghRepo: opts.ghRepo,
252
+ storage,
253
+ runDir,
254
+ maxConcurrency,
255
+ cellPlacement: opts.cellPlacement,
256
+ dispatchTimeoutMs: opts.dispatchTimeoutMs,
257
+ costLedger,
258
+ expectUsage,
259
+ labeledStore: opts.labeledStore,
260
+ captureSource: opts.captureSource,
261
+ analyzeGeneration: opts.analyzeGeneration,
262
+ findings: opts.findings,
263
+ selectionRankKey: opts.selectionRankKey
264
+ });
265
+ const reportSplit = holdoutDeferred ? "search" : "holdout";
266
+ const reportBaselineCampaign = holdoutDeferred ? result.baselineCampaign : result.baselineOnHoldout;
267
+ const reportWinnerCampaign = holdoutDeferred ? winnerSearchCampaign(result) : result.winnerOnHoldout;
268
+ const baseline = meanComposite(reportBaselineCampaign.aggregates.byScenario);
269
+ const winnerStats = meanComposite(reportWinnerCampaign.aggregates.byScenario);
270
+ let power;
271
+ const baselineHoldoutComposites = result.baselineOnHoldout.cells.filter((cell) => !cell.error).map((cell) => {
272
+ const scores = Object.values(cell.judgeScores);
273
+ return scores.length === 0 ? NaN : scores.reduce((sum, s) => sum + s.composite, 0) / scores.length;
274
+ }).filter((v) => Number.isFinite(v));
275
+ if (baselineHoldoutComposites.length >= 3) {
276
+ power = powerPreflight({
277
+ baselineComposites: baselineHoldoutComposites,
278
+ sharedScorerChannel: true
279
+ });
280
+ if (opts.onProgress) opts.onProgress({
281
+ kind: "power.estimated",
282
+ n: power.n,
283
+ sd: power.sd,
284
+ mde: power.mde,
285
+ underpowered: power.underpowered
286
+ });
287
+ if (power.underpowered && generations > 0) console.warn(`[selfImprove] ${power.recommendation}`);
288
+ }
289
+ if (opts.onProgress) {
290
+ opts.onProgress({
291
+ kind: "baseline.completed",
292
+ compositeMean: baseline.compositeMean,
293
+ durationMs: Date.now() - startedAt
294
+ });
295
+ opts.onProgress({
296
+ kind: "gate.decided",
297
+ decision: result.gateResult.decision,
298
+ ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean }
299
+ });
300
+ }
301
+ const cost = result.cost;
302
+ const totalCost = cost.totalCostUsd;
303
+ const insight = await analyzeRuns({
304
+ runs: [...cellsToRunRecords(reportBaselineCampaign.cells, "baseline", runDir, opts.baselineSurface, reportSplit, opts.model), ...cellsToRunRecords(reportWinnerCampaign.cells, "winner", runDir, result.winnerSurface, reportSplit, opts.model)],
305
+ baselineCandidateId: "baseline",
306
+ candidateCandidateId: "winner"
307
+ });
308
+ const durationMs = Date.now() - startedAt;
309
+ const { record: provenance } = await emitLoopProvenance({
310
+ ...loopProvenanceArgsFromResult({
311
+ runId: `${runDir}#${startedAt}`,
312
+ runDir,
313
+ timestamp: new Date(startedAt).toISOString(),
314
+ baselineSurface: opts.baselineSurface,
315
+ result,
316
+ costReceipts: costLedger.list(),
317
+ totalCostUsd: totalCost,
318
+ totalDurationMs: durationMs
319
+ }),
320
+ ...optimizationResult ? { optimizationMethod: {
321
+ name: opts.method.name,
322
+ cost: structuredClone(optimizationResult.cost),
323
+ ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
324
+ ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
325
+ } } : {},
326
+ storage,
327
+ hostedClient: opts.hostedTenant ? createHostedClient(opts.hostedTenant) : void 0
328
+ });
329
+ if (opts.onProvenance) opts.onProvenance(provenance);
330
+ const summary = {
331
+ baseline,
332
+ winner: {
333
+ ...winnerStats,
334
+ surface: result.winnerSurface,
335
+ ...result.winnerLabel ? { label: result.winnerLabel } : {},
336
+ ...result.winnerRationale ? { rationale: result.winnerRationale } : {}
337
+ },
338
+ ...holdoutDeferred ? {} : { lift: winnerStats.compositeMean - baseline.compositeMean },
339
+ diff: result.promotedDiff,
340
+ provenance,
341
+ gateDecision: result.gateResult.decision,
342
+ generationsExplored: result.generations.length,
343
+ durationMs,
344
+ totalCostUsd: totalCost,
345
+ cost,
346
+ receipts: costLedger.list(),
347
+ ...optimizationResult ? { optimization: {
348
+ name: opts.method.name,
349
+ cost: structuredClone(optimizationResult.cost),
350
+ ...optimizationResult.durationMs === void 0 ? {} : { durationMs: optimizationResult.durationMs },
351
+ ...optimizationResult.provenance === void 0 ? {} : { provenance: structuredClone(optimizationResult.provenance) }
352
+ } } : {},
353
+ insight,
354
+ ...power ? { power } : {},
355
+ raw: result
356
+ };
357
+ if (opts.hostedTenant) try {
358
+ await shipEvalRunToHosted(opts.hostedTenant, opts, summary, result, runDir);
359
+ } catch (err) {
360
+ const msg = err instanceof Error ? err.message : String(err);
361
+ console.warn(`[agent-eval] hosted ingest failed (continuing): ${msg}`);
362
+ }
363
+ return summary;
489
364
  }
490
365
  async function shipEvalRunToHosted(tenant, opts, summary, raw, runDir) {
491
- const client = createHostedClient(tenant);
492
- function snapshotFromCampaign(index, surface, campaign, durationMs) {
493
- const cells = campaign.cells.map((cell) => {
494
- const execution = campaignCellExecutionEvidence(cell);
495
- return {
496
- scenarioId: cell.scenarioId,
497
- rep: cell.rep,
498
- compositeMean: campaignCellTaskScore(cell) ?? null,
499
- dimensions: campaignCellJudgeDimensions(cell),
500
- terminalOutcome: execution.terminalOutcome,
501
- executionErrorCount: execution.executionErrorCount ?? null,
502
- errorMessage: cell.error ?? void 0
503
- };
504
- });
505
- const scoredCells = cells.flatMap(
506
- (cell) => cell.compositeMean === null ? [] : [cell.compositeMean]
507
- );
508
- const compositeMean = scoredCells.length === 0 ? null : scoredCells.reduce((sum, score) => sum + score, 0) / scoredCells.length;
509
- return {
510
- index,
511
- surfaceHash: surfaceHash(surface),
512
- surface,
513
- cells,
514
- compositeMean,
515
- costUsd: campaign.aggregates.totalCostUsd,
516
- durationMs
517
- };
518
- }
519
- const generations = [];
520
- generations.push(snapshotFromCampaign(0, opts.baselineSurface, raw.baselineCampaign, 0));
521
- for (const gen of raw.generations) {
522
- const winner = gen.surfaces.reduce(
523
- (best, s) => s.campaign.aggregates.cellsExecuted > 0 && (best === void 0 || averageComposite(s.campaign) > averageComposite(best.campaign)) ? s : best,
524
- gen.surfaces[0]
525
- );
526
- if (!winner) continue;
527
- generations.push(
528
- snapshotFromCampaign(gen.record.generationIndex + 1, winner.surface, winner.campaign, 0)
529
- );
530
- }
531
- const event = {
532
- runId: `${runDir}#${Date.now()}`,
533
- runDir,
534
- timestamp: (/* @__PURE__ */ new Date()).toISOString(),
535
- status: "finished",
536
- labels: opts.hostedLabels ?? {},
537
- baseline: generations[0],
538
- generations,
539
- gateDecision: summary.gateDecision,
540
- holdoutLift: summary.lift,
541
- totalCostUsd: summary.totalCostUsd,
542
- totalDurationMs: summary.durationMs,
543
- insightReport: summary.insight
544
- };
545
- await client.ingestEvalRun(event);
366
+ const client = createHostedClient(tenant);
367
+ function snapshotFromCampaign(index, surface, campaign, durationMs) {
368
+ const cells = campaign.cells.map((cell) => {
369
+ const execution = campaignCellExecutionEvidence(cell);
370
+ return {
371
+ scenarioId: cell.scenarioId,
372
+ rep: cell.rep,
373
+ compositeMean: campaignCellTaskScore(cell) ?? null,
374
+ dimensions: campaignCellJudgeDimensions(cell),
375
+ terminalOutcome: execution.terminalOutcome,
376
+ executionErrorCount: execution.executionErrorCount ?? null,
377
+ errorMessage: cell.error ?? void 0
378
+ };
379
+ });
380
+ const scoredCells = cells.flatMap((cell) => cell.compositeMean === null ? [] : [cell.compositeMean]);
381
+ const compositeMean = scoredCells.length === 0 ? null : scoredCells.reduce((sum, score) => sum + score, 0) / scoredCells.length;
382
+ return {
383
+ index,
384
+ surfaceHash: surfaceHash(surface),
385
+ surface,
386
+ cells,
387
+ compositeMean,
388
+ costUsd: campaign.aggregates.cost.totalCostUsd,
389
+ durationMs
390
+ };
391
+ }
392
+ const generations = [];
393
+ generations.push(snapshotFromCampaign(0, opts.baselineSurface, raw.baselineCampaign, 0));
394
+ for (const gen of raw.generations) {
395
+ const winner = gen.surfaces.reduce((best, s) => s.campaign.aggregates.cellsExecuted > 0 && (best === void 0 || averageComposite(s.campaign) > averageComposite(best.campaign)) ? s : best, gen.surfaces[0]);
396
+ if (!winner) continue;
397
+ generations.push(snapshotFromCampaign(gen.record.generationIndex + 1, winner.surface, winner.campaign, 0));
398
+ }
399
+ const event = {
400
+ runId: `${runDir}#${Date.now()}`,
401
+ runDir,
402
+ timestamp: (/* @__PURE__ */ new Date()).toISOString(),
403
+ status: "finished",
404
+ labels: opts.hostedLabels ?? {},
405
+ baseline: generations[0],
406
+ generations,
407
+ gateDecision: summary.gateDecision,
408
+ holdoutLift: summary.lift,
409
+ totalCostUsd: summary.totalCostUsd,
410
+ totalDurationMs: summary.durationMs,
411
+ insightReport: summary.insight
412
+ };
413
+ await client.ingestEvalRun(event);
546
414
  }
547
415
  function averageComposite(campaign) {
548
- const aggs = Object.values(campaign.aggregates.byScenario);
549
- return aggs.length === 0 ? 0 : aggs.reduce((s, a) => s + a.meanComposite, 0) / aggs.length;
416
+ const aggs = Object.values(campaign.aggregates.byScenario);
417
+ return aggs.length === 0 ? 0 : aggs.reduce((s, a) => s + a.meanComposite, 0) / aggs.length;
550
418
  }
551
419
  function hashString(s) {
552
- let h = 2166136261 >>> 0;
553
- for (let i = 0; i < s.length; i++) {
554
- h ^= s.charCodeAt(i);
555
- h = Math.imul(h, 16777619) >>> 0;
556
- }
557
- return h.toString(16).padStart(8, "0");
558
- }
420
+ let h = 2166136261;
421
+ for (let i = 0; i < s.length; i++) {
422
+ h ^= s.charCodeAt(i);
423
+ h = Math.imul(h, 16777619) >>> 0;
424
+ }
425
+ return h.toString(16).padStart(8, "0");
426
+ }
427
+ /**
428
+ * Adapt campaign cells into the `RunRecord` shape `analyzeRuns()` consumes.
429
+ * Each cell becomes one run; `candidateId` is the caller-supplied label so
430
+ * baseline + winner pair cleanly on `(experimentId, scenarioId, seed)`.
431
+ *
432
+ * `promptHash` is the REAL sha256 content hash of the surface this cell ran
433
+ * (baseline vs winner are byte-distinguishable + byte-identical-verifiable);
434
+ * `configHash` is the sha256 of the candidate label so the two candidates'
435
+ * config rows differ. Both were previously the literal `'sha256:cell'`, which
436
+ * made baseline and winner indistinguishable in every downstream record.
437
+ */
559
438
  function cellsToRunRecords(cells, candidateId, runId, surface, splitTag, fallbackModel) {
560
- const promptHash = surfaceContentHash(surface);
561
- const configHash = surfaceContentHash(candidateId);
562
- return cells.map((cell) => {
563
- const model = cell.resolvedModel ?? fallbackModel;
564
- if (!model) {
565
- throw new ValidationError(
566
- `selfImprove.model is required when cell ${cell.cellId} has no paid-call model receipt`
567
- );
568
- }
569
- if (!modelHasSnapshot(model)) {
570
- throw new ValidationError(
571
- `selfImprove model "${model}" lacks a snapshot version for cell ${cell.cellId}`
572
- );
573
- }
574
- return campaignCellToRunRecord(cell, {
575
- runId: `${runId}::${candidateId}::${cell.cellId}`,
576
- experimentId: runId,
577
- candidateId,
578
- // scenarioId is explicit; seed keeps repeated runs distinct.
579
- seed: cell.rep * 1e6 + hashString(cell.scenarioId).slice(0, 6).split("").reduce((a, c) => a * 31 + c.charCodeAt(0) >>> 0, 0),
580
- model,
581
- promptHash,
582
- configHash,
583
- commitSha: "cell",
584
- splitTag
585
- });
586
- });
587
- }
588
-
589
- // src/contract/define-agent-eval.ts
439
+ const promptHash = surfaceContentHash(surface);
440
+ const configHash = surfaceContentHash(candidateId);
441
+ return cells.map((cell) => {
442
+ const model = cell.resolvedModel ?? fallbackModel;
443
+ if (!model) throw new ValidationError(`selfImprove.model is required when cell ${cell.cellId} has no paid-call model receipt`);
444
+ if (!modelHasSnapshot(model)) throw new ValidationError(`selfImprove model "${model}" lacks a snapshot version for cell ${cell.cellId}`);
445
+ return campaignCellToRunRecord(cell, {
446
+ runId: `${runId}::${candidateId}::${cell.cellId}`,
447
+ experimentId: runId,
448
+ candidateId,
449
+ seed: cell.rep * 1e6 + hashString(cell.scenarioId).slice(0, 6).split("").reduce((a, c) => a * 31 + c.charCodeAt(0) >>> 0, 0),
450
+ model,
451
+ promptHash,
452
+ configHash,
453
+ commitSha: "cell",
454
+ splitTag
455
+ });
456
+ });
457
+ }
458
+ //#endregion
459
+ //#region src/contract/define-agent-eval.ts
460
+ /**
461
+ * Define an agent eval once, then either score a surface with `evaluate()` or
462
+ * run the closed loop with `improve()`.
463
+ *
464
+ * This is a DX wrapper only: it delegates to `runEval()` and `selfImprove()` and
465
+ * returns their native result shapes.
466
+ */
590
467
  function defineAgentEval(defaults) {
591
- const defaultEvaluateOptions = evaluateDefaults(defaults);
592
- return {
593
- scenarios: defaults.scenarios,
594
- baselineSurface: defaults.baselineSurface,
595
- async evaluate(opts = {}) {
596
- const { agent, judge, judges, runDir, scenarios, surface, ...campaignOpts } = opts;
597
- const selectedAgent = agent ?? defaults.agent;
598
- const selectedSurface = surface ?? defaults.baselineSurface;
599
- const selectedRunDir = runDir ?? defaults.runDir ?? `mem://defineAgentEval-${Date.now()}`;
600
- const selectedStorage = campaignOpts.storage ?? defaultEvaluateOptions.storage ?? (selectedRunDir.startsWith("mem://") ? inMemoryCampaignStorage() : void 0);
601
- const evalOptions = {
602
- ...defaultEvaluateOptions,
603
- ...campaignOpts,
604
- ...selectedStorage ? { storage: selectedStorage } : {},
605
- runDir: selectedRunDir,
606
- scenarios: scenarios ?? defaults.scenarios,
607
- dispatch: (scenario, ctx) => selectedAgent(selectedSurface, scenario, ctx),
608
- judges: evaluateJudges(judges, judge ?? defaults.judge)
609
- };
610
- if (evalOptions.reps !== void 0)
611
- evalOptions.reps = requirePositiveInteger(evalOptions.reps, "reps");
612
- return runEval(evalOptions);
613
- },
614
- async improve(opts = {}) {
615
- const {
616
- budget: budgetOverride,
617
- hostedTenant: hostedTenantOverride,
618
- ...topLevelOverrides
619
- } = opts;
620
- const merged = mergeDefined(defaults, topLevelOverrides);
621
- const budget = mergeBudget(defaults.budget, budgetOverride);
622
- const hostedTenant = mergeHostedTenant(defaults.hostedTenant, hostedTenantOverride);
623
- return selfImprove({
624
- ...merged,
625
- ...budget ? { budget } : {},
626
- ...hostedTenant ? { hostedTenant } : {}
627
- });
628
- }
629
- };
468
+ const defaultEvaluateOptions = evaluateDefaults(defaults);
469
+ return {
470
+ scenarios: defaults.scenarios,
471
+ baselineSurface: defaults.baselineSurface,
472
+ async evaluate(opts = {}) {
473
+ const { agent, judge, judges, runDir, scenarios, surface, ...campaignOpts } = opts;
474
+ const selectedAgent = agent ?? defaults.agent;
475
+ const selectedSurface = surface ?? defaults.baselineSurface;
476
+ const selectedRunDir = runDir ?? defaults.runDir ?? `mem://defineAgentEval-${Date.now()}`;
477
+ const selectedStorage = campaignOpts.storage ?? defaultEvaluateOptions.storage ?? (selectedRunDir.startsWith("mem://") ? inMemoryCampaignStorage() : void 0);
478
+ const evalOptions = {
479
+ ...defaultEvaluateOptions,
480
+ ...campaignOpts,
481
+ ...selectedStorage ? { storage: selectedStorage } : {},
482
+ runDir: selectedRunDir,
483
+ scenarios: scenarios ?? defaults.scenarios,
484
+ dispatch: (scenario, ctx) => selectedAgent(selectedSurface, scenario, ctx),
485
+ judges: evaluateJudges(judges, judge ?? defaults.judge)
486
+ };
487
+ if (evalOptions.reps !== void 0) evalOptions.reps = requirePositiveInteger(evalOptions.reps, "reps");
488
+ return runEval(evalOptions);
489
+ },
490
+ async improve(opts = {}) {
491
+ const { budget: budgetOverride, hostedTenant: hostedTenantOverride, ...topLevelOverrides } = opts;
492
+ const merged = mergeDefined(defaults, topLevelOverrides);
493
+ const budget = mergeBudget(defaults.budget, budgetOverride);
494
+ const hostedTenant = mergeHostedTenant(defaults.hostedTenant, hostedTenantOverride);
495
+ return selfImprove({
496
+ ...merged,
497
+ ...budget ? { budget } : {},
498
+ ...hostedTenant ? { hostedTenant } : {}
499
+ });
500
+ }
501
+ };
630
502
  }
631
503
  function evaluateDefaults(defaults) {
632
- const out = {};
633
- if (defaults.storage) out.storage = defaults.storage;
634
- if (defaults.labeledStore) out.labeledStore = defaults.labeledStore;
635
- if (defaults.captureSource) out.captureSource = defaults.captureSource;
636
- if (defaults.cellPlacement) out.cellPlacement = defaults.cellPlacement;
637
- if (defaults.expectUsage) out.expectUsage = defaults.expectUsage;
638
- if (defaults.budget?.dollars !== void 0) out.costCeiling = defaults.budget.dollars;
639
- if (defaults.budget?.maxConcurrency !== void 0)
640
- out.maxConcurrency = defaults.budget.maxConcurrency;
641
- if (defaults.budget?.reps !== void 0)
642
- out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
643
- return out;
504
+ const out = {};
505
+ if (defaults.storage) out.storage = defaults.storage;
506
+ if (defaults.labeledStore) out.labeledStore = defaults.labeledStore;
507
+ if (defaults.captureSource) out.captureSource = defaults.captureSource;
508
+ if (defaults.cellPlacement) out.cellPlacement = defaults.cellPlacement;
509
+ if (defaults.expectUsage) out.expectUsage = defaults.expectUsage;
510
+ if (defaults.budget?.dollars !== void 0) out.costCeiling = defaults.budget.dollars;
511
+ if (defaults.budget?.maxConcurrency !== void 0) out.maxConcurrency = defaults.budget.maxConcurrency;
512
+ if (defaults.budget?.reps !== void 0) out.reps = requirePositiveInteger(defaults.budget.reps, "budget.reps");
513
+ return out;
644
514
  }
645
515
  function mergeBudget(defaults, overrides) {
646
- const merged = mergeOptionalObject(defaults, overrides);
647
- if (merged?.reps !== void 0) merged.reps = requirePositiveInteger(merged.reps, "budget.reps");
648
- return merged;
516
+ const merged = mergeOptionalObject(defaults, overrides);
517
+ if (merged?.reps !== void 0) merged.reps = requirePositiveInteger(merged.reps, "budget.reps");
518
+ return merged;
649
519
  }
650
520
  function mergeHostedTenant(defaults, overrides) {
651
- const merged = mergeOptionalObject(defaults, overrides);
652
- if (!merged) return void 0;
653
- if (!merged.endpoint?.trim() || !merged.apiKey?.trim() || !merged.tenantId?.trim()) {
654
- throw new Error(
655
- "defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides"
656
- );
657
- }
658
- return merged;
521
+ const merged = mergeOptionalObject(defaults, overrides);
522
+ if (!merged) return void 0;
523
+ if (!merged.endpoint?.trim() || !merged.apiKey?.trim() || !merged.tenantId?.trim()) throw new Error("defineAgentEval.improve: hostedTenant requires endpoint, apiKey, and tenantId after merging defaults and overrides");
524
+ return merged;
659
525
  }
660
526
  function mergeDefined(defaults, overrides) {
661
- if (!overrides) return defaults;
662
- const merged = { ...defaults };
663
- for (const [key, value] of Object.entries(overrides)) {
664
- if (value !== void 0) merged[key] = value;
665
- }
666
- return merged;
527
+ if (!overrides) return defaults;
528
+ const merged = { ...defaults };
529
+ for (const [key, value] of Object.entries(overrides)) if (value !== void 0) merged[key] = value;
530
+ return merged;
667
531
  }
668
532
  function mergeOptionalObject(defaults, overrides) {
669
- if (!defaults && !overrides) return void 0;
670
- return mergeDefined(defaults ?? {}, overrides);
533
+ if (!defaults && !overrides) return void 0;
534
+ return mergeDefined(defaults ?? {}, overrides);
671
535
  }
672
536
  function evaluateJudges(judges, defaultJudge) {
673
- if (judges !== void 0) {
674
- if (judges.length === 0) {
675
- throw new Error("defineAgentEval.evaluate: judges must not be empty");
676
- }
677
- return judges;
678
- }
679
- return [defaultJudge];
537
+ if (judges !== void 0) {
538
+ if (judges.length === 0) throw new Error("defineAgentEval.evaluate: judges must not be empty");
539
+ return judges;
540
+ }
541
+ return [defaultJudge];
680
542
  }
681
543
  function requirePositiveInteger(value, field) {
682
- if (!Number.isInteger(value) || value < 1) {
683
- throw new Error(`defineAgentEval: ${field} must be a positive integer`);
684
- }
685
- return value;
544
+ if (!Number.isInteger(value) || value < 1) throw new Error(`defineAgentEval: ${field} must be a positive integer`);
545
+ return value;
686
546
  }
687
-
688
- // src/contract/measured-comparison.ts
689
- import {
690
- agentCandidateBenchmarkSuiteSchema,
691
- agentCandidateBenchmarkTaskSchema,
692
- agentCandidateBundleSchema,
693
- agentCandidateEvaluationPolicySchema,
694
- agentCandidateExperimentSchema,
695
- agentImprovementMeasuredComparisonSchema,
696
- candidateExecutionEvidenceSchema,
697
- canonicalCandidateDigest,
698
- omitTopLevelDigest
699
- } from "@tangle-network/agent-interface";
547
+ //#endregion
548
+ //#region src/contract/measured-comparison.ts
549
+ /** Content-address one task before any measured execution can see it. */
700
550
  function sealCandidateBenchmarkTask(material) {
701
- return agentCandidateBenchmarkTaskSchema.parse({
702
- ...material,
703
- digest: canonicalCandidateDigest(material)
704
- });
551
+ return agentCandidateBenchmarkTaskSchema.parse({
552
+ ...material,
553
+ digest: canonicalCandidateDigest(material)
554
+ });
705
555
  }
556
+ /** Freeze task order, repetitions, and every seed before either arm runs. */
706
557
  function sealCandidateBenchmarkSuite(options) {
707
- for (const task of options.tasks) verifyCandidateBenchmarkTask(task);
708
- const material = {
709
- kind: "agent-candidate-benchmark-suite",
710
- digestAlgorithm: "rfc8785-sha256",
711
- taskDigests: options.tasks.map((task) => task.digest),
712
- reps: options.reps,
713
- seeds: options.seeds
714
- };
715
- const suite = agentCandidateBenchmarkSuiteSchema.parse({
716
- ...material,
717
- digest: canonicalCandidateDigest(material)
718
- });
719
- return { suite, tasks: options.tasks };
720
- }
558
+ for (const task of options.tasks) verifyCandidateBenchmarkTask(task);
559
+ const material = {
560
+ kind: "agent-candidate-benchmark-suite",
561
+ digestAlgorithm: "rfc8785-sha256",
562
+ taskDigests: options.tasks.map((task) => task.digest),
563
+ reps: options.reps,
564
+ seeds: options.seeds
565
+ };
566
+ return {
567
+ suite: agentCandidateBenchmarkSuiteSchema.parse({
568
+ ...material,
569
+ digest: canonicalCandidateDigest(material)
570
+ }),
571
+ tasks: options.tasks
572
+ };
573
+ }
574
+ /** Freeze both complete agent states and their exact held-out work. */
721
575
  function sealCandidateExperiment(material) {
722
- const parsed = agentCandidateExperimentSchema.parse({
723
- ...material,
724
- digest: canonicalCandidateDigest(material)
725
- });
726
- return verifyCandidateExperiment(parsed);
576
+ return verifyCandidateExperiment(agentCandidateExperimentSchema.parse({
577
+ ...material,
578
+ digest: canonicalCandidateDigest(material)
579
+ }));
727
580
  }
728
581
  function verifyCandidateExperiment(input) {
729
- const experiment = agentCandidateExperimentSchema.parse(input);
730
- verifySelfAddressed(experiment, "candidate experiment");
731
- verifyBundle(experiment.baseline, "baseline bundle");
732
- verifyBundle(experiment.candidate, "candidate bundle");
733
- if (experiment.baseline.digest === experiment.candidate.digest) {
734
- throw new Error("candidate experiment baseline and candidate bundles are identical");
735
- }
736
- verifyCandidateBenchmarkSuiteInputs(experiment.benchmark);
737
- return experiment;
738
- }
582
+ const experiment = agentCandidateExperimentSchema.parse(input);
583
+ verifySelfAddressed(experiment, "candidate experiment");
584
+ verifyBundle(experiment.baseline, "baseline bundle");
585
+ verifyBundle(experiment.candidate, "candidate bundle");
586
+ if (experiment.baseline.digest === experiment.candidate.digest) throw new Error("candidate experiment baseline and candidate bundles are identical");
587
+ verifyCandidateBenchmarkSuiteInputs(experiment.benchmark);
588
+ return experiment;
589
+ }
590
+ /** Execute each signed cell for both arms. The callback is Runtime's one executor. */
739
591
  async function runCandidateExperiment(options) {
740
- const experiment = verifyCandidateExperiment(options.experiment);
741
- const { suite, tasks } = experiment.benchmark;
742
- const maxConcurrency = options.maxConcurrency ?? 2;
743
- if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) {
744
- throw new Error("candidate experiment maxConcurrency must be a positive integer");
745
- }
746
- const measurements = new Array(
747
- suite.taskDigests.length * suite.reps
748
- );
749
- let nextIndex = 0;
750
- const lanes = Array.from({ length: Math.min(maxConcurrency, measurements.length) }, async () => {
751
- while (true) {
752
- if (options.signal?.aborted) throw abortError(options.signal);
753
- const index = nextIndex;
754
- nextIndex += 1;
755
- if (index >= measurements.length) return;
756
- const taskIndex = Math.floor(index / suite.reps);
757
- const repetition = index % suite.reps;
758
- const task = tasks[taskIndex];
759
- const seed = suite.seeds[index];
760
- if (!task || seed === void 0) {
761
- throw new Error(`candidate experiment cell ${index} has no signed task or seed`);
762
- }
763
- const benchmarkCell = {
764
- suiteDigest: suite.digest,
765
- taskIndex,
766
- repetition
767
- };
768
- const [baseline, candidate] = await Promise.all([
769
- options.execute({
770
- experiment,
771
- arm: "baseline",
772
- bundle: experiment.baseline,
773
- task,
774
- benchmarkCell,
775
- seed,
776
- ...options.signal ? { signal: options.signal } : {}
777
- }),
778
- options.execute({
779
- experiment,
780
- arm: "candidate",
781
- bundle: experiment.candidate,
782
- task,
783
- benchmarkCell,
784
- seed,
785
- ...options.signal ? { signal: options.signal } : {}
786
- })
787
- ]);
788
- const measurement = { baseline, candidate };
789
- verifyMeasurement(experiment, measurement, index);
790
- measurements[index] = measurement;
791
- }
792
- });
793
- await Promise.all(lanes);
794
- return measurements;
795
- }
592
+ const experiment = verifyCandidateExperiment(options.experiment);
593
+ const { suite, tasks } = experiment.benchmark;
594
+ const maxConcurrency = options.maxConcurrency ?? 2;
595
+ if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) throw new Error("candidate experiment maxConcurrency must be a positive integer");
596
+ const measurements = new Array(suite.taskDigests.length * suite.reps);
597
+ let nextIndex = 0;
598
+ const lanes = Array.from({ length: Math.min(maxConcurrency, measurements.length) }, async () => {
599
+ while (true) {
600
+ if (options.signal?.aborted) throw abortError(options.signal);
601
+ const index = nextIndex;
602
+ nextIndex += 1;
603
+ if (index >= measurements.length) return;
604
+ const taskIndex = Math.floor(index / suite.reps);
605
+ const repetition = index % suite.reps;
606
+ const task = tasks[taskIndex];
607
+ const seed = suite.seeds[index];
608
+ if (!task || seed === void 0) throw new Error(`candidate experiment cell ${index} has no signed task or seed`);
609
+ const benchmarkCell = {
610
+ suiteDigest: suite.digest,
611
+ taskIndex,
612
+ repetition
613
+ };
614
+ const [baseline, candidate] = await Promise.all([options.execute({
615
+ experiment,
616
+ arm: "baseline",
617
+ bundle: experiment.baseline,
618
+ task,
619
+ benchmarkCell,
620
+ seed,
621
+ ...options.signal ? { signal: options.signal } : {}
622
+ }), options.execute({
623
+ experiment,
624
+ arm: "candidate",
625
+ bundle: experiment.candidate,
626
+ task,
627
+ benchmarkCell,
628
+ seed,
629
+ ...options.signal ? { signal: options.signal } : {}
630
+ })]);
631
+ const measurement = {
632
+ baseline,
633
+ candidate
634
+ };
635
+ verifyMeasurement(experiment, measurement, index);
636
+ measurements[index] = measurement;
637
+ }
638
+ });
639
+ await Promise.all(lanes);
640
+ return measurements;
641
+ }
642
+ /**
643
+ * Calculate the shared paired decision from any complete receipt shape.
644
+ *
645
+ * Callers still own sealing their tasks, verifying each receipt against its
646
+ * expected arm and state, and proving every expected cell exists. This function
647
+ * only validates the projected measurements and derives their shared decision.
648
+ */
796
649
  function evaluatePairedMeasurements(options) {
797
- if (options.measurements.length === 0) {
798
- throw new Error("paired measurement evaluation requires at least one paired cell");
799
- }
800
- const additionalCostUsd = options.additionalCostUsd ?? 0;
801
- if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) {
802
- throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
803
- }
804
- if (typeof options.sharedScorerChannel !== "boolean") {
805
- throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
806
- }
807
- const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
808
- const measurements = options.measurements.map(
809
- (measurement, index) => projectPairedMeasurement(measurement, index, options.adapter)
810
- );
811
- const cellIds2 = measurements.map((measurement) => measurement.cellId);
812
- if (new Set(cellIds2).size !== cellIds2.length) {
813
- throw new Error("paired measurement evaluation cell ids must be unique");
814
- }
815
- const dimensions = sharedProjectedDimensions(measurements);
816
- const baselineScores = measurements.map((measurement) => measurement.baseline.score);
817
- const candidateScores = measurements.map((measurement) => measurement.candidate.score);
818
- const {
819
- confidenceLevel: confidence,
820
- resamples,
821
- bootstrapSeed,
822
- deltaThreshold,
823
- minProductiveRuns,
824
- budgetUsd,
825
- criticalDimensions,
826
- regressionTolerance
827
- } = policy;
828
- const significance = heldoutSignificance(
829
- { before: baselineScores, after: candidateScores, cellIds: cellIds2 },
830
- {
831
- confidence,
832
- resamples,
833
- seed: bootstrapSeed,
834
- statistic: "mean",
835
- deltaThreshold,
836
- minProductiveRuns
837
- }
838
- );
839
- const overall = measuredEstimate(baselineScores, candidateScores, {
840
- confidence,
841
- resamples,
842
- seed: bootstrapSeed
843
- });
844
- const objectives = [
845
- {
846
- kind: "objective",
847
- name: "benchmark-score",
848
- direction: "higher-is-better",
849
- unit: "score",
850
- availability: "measured",
851
- ...overall
852
- },
853
- ...dimensions.map((name, index) => ({
854
- kind: "dimension",
855
- objective: "benchmark-score",
856
- name,
857
- direction: "higher-is-better",
858
- unit: "score",
859
- availability: "measured",
860
- ...measuredEstimate(
861
- measurements.map((measurement) => dimensionScore(measurement.baseline, name)),
862
- measurements.map((measurement) => dimensionScore(measurement.candidate, name)),
863
- {
864
- confidence,
865
- resamples,
866
- seed: bootstrapSeed + index + 1
867
- }
868
- )
869
- }))
870
- ];
871
- const cost = measuredEstimate(
872
- measurements.map((measurement) => measurement.baseline.costUsd),
873
- measurements.map((measurement) => measurement.candidate.costUsd),
874
- {
875
- confidence,
876
- resamples,
877
- seed: bootstrapSeed + dimensions.length + 1
878
- }
879
- );
880
- const latency = measuredEstimate(
881
- measurements.map((measurement) => measurement.baseline.latencyMs),
882
- measurements.map((measurement) => measurement.candidate.latencyMs),
883
- {
884
- confidence,
885
- resamples,
886
- seed: bootstrapSeed + dimensions.length + 2
887
- }
888
- );
889
- objectives.push(
890
- {
891
- kind: "cost",
892
- name: "cost",
893
- direction: "lower-is-better",
894
- unit: "usd",
895
- availability: "measured",
896
- ...cost
897
- },
898
- {
899
- kind: "latency",
900
- name: "latency",
901
- direction: "lower-is-better",
902
- unit: "milliseconds",
903
- availability: "measured",
904
- ...latency
905
- }
906
- );
907
- const power = baselineScores.length >= 3 ? powerPreflight({
908
- baselineComposites: baselineScores,
909
- pairedN: baselineScores.length,
910
- deltaThreshold,
911
- confidence,
912
- sharedScorerChannel: options.sharedScorerChannel
913
- }) : void 0;
914
- const powerSufficient = baselineScores.length >= minProductiveRuns && power !== void 0 && !power.underpowered;
915
- const guardedDimensions = new Set(criticalDimensions);
916
- const missingCriticalDimensions = criticalDimensions.filter(
917
- (dimension) => !dimensions.includes(dimension)
918
- );
919
- const regressions = objectives.filter(
920
- (objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance
921
- );
922
- const executionCostUsd = measurements.reduce(
923
- (sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd,
924
- 0
925
- );
926
- const executionDurationMs = measurements.reduce(
927
- (sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs,
928
- 0
929
- );
930
- const completedRuns = measurements.flatMap((measurement) => [
931
- measurement.baseline,
932
- measurement.candidate
933
- ]);
934
- const incompleteRuns = completedRuns.filter((run) => !run.completed);
935
- const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
936
- const totalCostUsd = executionCostUsd + additionalCostUsd;
937
- const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
938
- const checks = [
939
- { name: "paired-significance", passed: significance.significant },
940
- { name: "statistical-power", passed: powerSufficient },
941
- { name: "all-runs-completed", passed: incompleteRuns.length === 0 },
942
- { name: "candidate-task-pass", passed: failedCandidateResults.length === 0 },
943
- {
944
- name: "critical-dimensions",
945
- passed: regressions.length === 0 && missingCriticalDimensions.length === 0
946
- },
947
- { name: "budget", passed: budgetPassed }
948
- ];
949
- const shipped = checks.every((check) => check.passed);
950
- const reasons = [
951
- ...significance.significant ? [] : [
952
- significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`
953
- ],
954
- ...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
955
- ...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
956
- ...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
957
- ...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
958
- ...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
959
- ...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
960
- ];
961
- return {
962
- overall: {
963
- name: "composite",
964
- direction: "higher-is-better",
965
- unit: "score",
966
- ...overall
967
- },
968
- objectives,
969
- decision: {
970
- outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
971
- reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
972
- contributingChecks: checks
973
- },
974
- power: {
975
- sufficient: powerSufficient,
976
- n: baselineScores.length,
977
- minimumDetectableDelta: power?.mde ?? 1,
978
- confidenceLevel: confidence,
979
- scaleAssumed: power?.scaleAssumed ?? true,
980
- sharedScorerChannel: options.sharedScorerChannel,
981
- reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
982
- },
983
- executionCostUsd,
984
- totalCostUsd,
985
- executionDurationMs
986
- };
987
- }
650
+ if (options.measurements.length === 0) throw new Error("paired measurement evaluation requires at least one paired cell");
651
+ const additionalCostUsd = options.additionalCostUsd ?? 0;
652
+ if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
653
+ if (typeof options.sharedScorerChannel !== "boolean") throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
654
+ const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
655
+ const measurements = options.measurements.map((measurement, index) => projectPairedMeasurement(measurement, index, options.adapter));
656
+ const cellIds = measurements.map((measurement) => measurement.cellId);
657
+ if (new Set(cellIds).size !== cellIds.length) throw new Error("paired measurement evaluation cell ids must be unique");
658
+ const dimensions = sharedProjectedDimensions(measurements);
659
+ const baselineScores = measurements.map((measurement) => measurement.baseline.score);
660
+ const candidateScores = measurements.map((measurement) => measurement.candidate.score);
661
+ const { confidenceLevel: confidence, resamples, bootstrapSeed, deltaThreshold, minProductiveRuns, budgetUsd, criticalDimensions, regressionTolerance } = policy;
662
+ const significance = heldoutSignificance({
663
+ before: baselineScores,
664
+ after: candidateScores,
665
+ cellIds
666
+ }, {
667
+ confidence,
668
+ resamples,
669
+ seed: bootstrapSeed,
670
+ statistic: "mean",
671
+ deltaThreshold,
672
+ minProductiveRuns
673
+ });
674
+ const overall = measuredEstimate(baselineScores, candidateScores, {
675
+ confidence,
676
+ resamples,
677
+ seed: bootstrapSeed
678
+ });
679
+ const objectives = [{
680
+ kind: "objective",
681
+ name: "benchmark-score",
682
+ direction: "higher-is-better",
683
+ unit: "score",
684
+ availability: "measured",
685
+ ...overall
686
+ }, ...dimensions.map((name, index) => ({
687
+ kind: "dimension",
688
+ objective: "benchmark-score",
689
+ name,
690
+ direction: "higher-is-better",
691
+ unit: "score",
692
+ availability: "measured",
693
+ ...measuredEstimate(measurements.map((measurement) => dimensionScore(measurement.baseline, name)), measurements.map((measurement) => dimensionScore(measurement.candidate, name)), {
694
+ confidence,
695
+ resamples,
696
+ seed: bootstrapSeed + index + 1
697
+ })
698
+ }))];
699
+ const cost = measuredEstimate(measurements.map((measurement) => measurement.baseline.costUsd), measurements.map((measurement) => measurement.candidate.costUsd), {
700
+ confidence,
701
+ resamples,
702
+ seed: bootstrapSeed + dimensions.length + 1
703
+ });
704
+ const latency = measuredEstimate(measurements.map((measurement) => measurement.baseline.latencyMs), measurements.map((measurement) => measurement.candidate.latencyMs), {
705
+ confidence,
706
+ resamples,
707
+ seed: bootstrapSeed + dimensions.length + 2
708
+ });
709
+ objectives.push({
710
+ kind: "cost",
711
+ name: "cost",
712
+ direction: "lower-is-better",
713
+ unit: "usd",
714
+ availability: "measured",
715
+ ...cost
716
+ }, {
717
+ kind: "latency",
718
+ name: "latency",
719
+ direction: "lower-is-better",
720
+ unit: "milliseconds",
721
+ availability: "measured",
722
+ ...latency
723
+ });
724
+ const power = baselineScores.length >= 3 ? powerPreflight({
725
+ baselineComposites: baselineScores,
726
+ pairedN: baselineScores.length,
727
+ deltaThreshold,
728
+ confidence,
729
+ sharedScorerChannel: options.sharedScorerChannel
730
+ }) : void 0;
731
+ const powerSufficient = baselineScores.length >= minProductiveRuns && power !== void 0 && !power.underpowered;
732
+ const guardedDimensions = new Set(criticalDimensions);
733
+ const missingCriticalDimensions = criticalDimensions.filter((dimension) => !dimensions.includes(dimension));
734
+ const regressions = objectives.filter((objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance);
735
+ const executionCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
736
+ const executionDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
737
+ const incompleteRuns = measurements.flatMap((measurement) => [measurement.baseline, measurement.candidate]).filter((run) => !run.completed);
738
+ const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
739
+ const totalCostUsd = executionCostUsd + additionalCostUsd;
740
+ const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
741
+ const checks = [
742
+ {
743
+ name: "paired-significance",
744
+ passed: significance.significant
745
+ },
746
+ {
747
+ name: "statistical-power",
748
+ passed: powerSufficient
749
+ },
750
+ {
751
+ name: "all-runs-completed",
752
+ passed: incompleteRuns.length === 0
753
+ },
754
+ {
755
+ name: "candidate-task-pass",
756
+ passed: failedCandidateResults.length === 0
757
+ },
758
+ {
759
+ name: "critical-dimensions",
760
+ passed: regressions.length === 0 && missingCriticalDimensions.length === 0
761
+ },
762
+ {
763
+ name: "budget",
764
+ passed: budgetPassed
765
+ }
766
+ ];
767
+ const shipped = checks.every((check) => check.passed);
768
+ const reasons = [
769
+ ...significance.significant ? [] : [significance.fewRuns ? `only ${significance.n} paired runs; ${minProductiveRuns} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`],
770
+ ...powerSufficient ? [] : [power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`],
771
+ ...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
772
+ ...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
773
+ ...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
774
+ ...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
775
+ ...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
776
+ ];
777
+ return {
778
+ overall: {
779
+ name: "composite",
780
+ direction: "higher-is-better",
781
+ unit: "score",
782
+ ...overall
783
+ },
784
+ objectives,
785
+ decision: {
786
+ outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
787
+ reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
788
+ contributingChecks: checks
789
+ },
790
+ power: {
791
+ sufficient: powerSufficient,
792
+ n: baselineScores.length,
793
+ minimumDetectableDelta: power?.mde ?? 1,
794
+ confidenceLevel: confidence,
795
+ scaleAssumed: power?.scaleAssumed ?? true,
796
+ sharedScorerChannel: options.sharedScorerChannel,
797
+ reason: power?.recommendation ?? `need at least ${Math.max(3, minProductiveRuns)} paired runs`
798
+ },
799
+ executionCostUsd,
800
+ totalCostUsd,
801
+ executionDurationMs
802
+ };
803
+ }
804
+ /** Build the only publishable comparison: paired statistics over Runtime receipts. */
988
805
  function measuredComparisonFromCandidateExperiment(options) {
989
- const experiment = verifyCandidateExperiment(options.experiment);
990
- const measurements = options.measurements.map(
991
- (measurement, index) => verifyMeasurement(experiment, measurement, index)
992
- );
993
- const expectedN = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
994
- if (measurements.length !== expectedN) {
995
- throw new Error(
996
- `candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`
997
- );
998
- }
999
- verifyStableProfileMaterialization(measurements);
1000
- if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
1001
- const searchCostUsd = options.searchCostUsd ?? 0;
1002
- const evaluation = evaluatePairedMeasurements({
1003
- measurements: measurements.map((measurement, index) => ({
1004
- cellId: cellIds(experiment)[index],
1005
- ...measurement
1006
- })),
1007
- policy: experiment.policy,
1008
- adapter: candidateExecutionEvidenceAdapter,
1009
- sharedScorerChannel: true,
1010
- additionalCostUsd: searchCostUsd
1011
- });
1012
- const diff = deriveCandidateBundleDiff(experiment);
1013
- const searchDurationMs = options.searchDurationMs ?? 0;
1014
- const totalCostUsd = evaluation.totalCostUsd;
1015
- const durationMs = evaluation.executionDurationMs + searchDurationMs;
1016
- const provisional = agentImprovementMeasuredComparisonSchema.parse({
1017
- kind: "agent-improvement-measured-comparison",
1018
- experiment,
1019
- measurements,
1020
- overall: evaluation.overall,
1021
- objectives: evaluation.objectives,
1022
- ...options.candidate ? { candidate: options.candidate } : {},
1023
- decision: evaluation.decision,
1024
- power: evaluation.power,
1025
- provenance: {
1026
- kind: "agent-eval-loop",
1027
- schema: "agent-candidate-experiment",
1028
- runId: options.runId,
1029
- recordDigest: canonicalCandidateDigest({}),
1030
- baselineContentHash: experiment.baseline.digest,
1031
- candidateContentHash: experiment.candidate.digest
1032
- },
1033
- diff,
1034
- evaluation: {
1035
- generationsExplored: options.generationsExplored ?? 0,
1036
- searchDurationMs,
1037
- executionDurationMs: evaluation.executionDurationMs,
1038
- durationMs,
1039
- searchCostUsd,
1040
- executionCostUsd: evaluation.executionCostUsd,
1041
- totalCostUsd
1042
- },
1043
- ...options.metadata ? { metadata: options.metadata } : {}
1044
- });
1045
- const { recordDigest: _recordDigest, ...provenance } = provisional.provenance;
1046
- return agentImprovementMeasuredComparisonSchema.parse({
1047
- ...provisional,
1048
- provenance: {
1049
- ...provenance,
1050
- recordDigest: canonicalCandidateDigest({ ...provisional, provenance })
1051
- }
1052
- });
1053
- }
806
+ const experiment = verifyCandidateExperiment(options.experiment);
807
+ const measurements = options.measurements.map((measurement, index) => verifyMeasurement(experiment, measurement, index));
808
+ const expectedN = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
809
+ if (measurements.length !== expectedN) throw new Error(`candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`);
810
+ verifyStableProfileMaterialization(measurements);
811
+ if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
812
+ const searchCostUsd = options.searchCostUsd ?? 0;
813
+ const evaluation = evaluatePairedMeasurements({
814
+ measurements: measurements.map((measurement, index) => ({
815
+ cellId: cellIds(experiment)[index],
816
+ ...measurement
817
+ })),
818
+ policy: experiment.policy,
819
+ adapter: candidateExecutionEvidenceAdapter,
820
+ sharedScorerChannel: true,
821
+ additionalCostUsd: searchCostUsd
822
+ });
823
+ const diff = deriveCandidateBundleDiff(experiment);
824
+ const searchDurationMs = options.searchDurationMs ?? 0;
825
+ const totalCostUsd = evaluation.totalCostUsd;
826
+ const durationMs = evaluation.executionDurationMs + searchDurationMs;
827
+ const provisional = agentImprovementMeasuredComparisonSchema.parse({
828
+ kind: "agent-improvement-measured-comparison",
829
+ experiment,
830
+ measurements,
831
+ overall: evaluation.overall,
832
+ objectives: evaluation.objectives,
833
+ ...options.candidate ? { candidate: options.candidate } : {},
834
+ decision: evaluation.decision,
835
+ power: evaluation.power,
836
+ provenance: {
837
+ kind: "agent-eval-loop",
838
+ schema: "agent-candidate-experiment",
839
+ runId: options.runId,
840
+ recordDigest: canonicalCandidateDigest({}),
841
+ baselineContentHash: experiment.baseline.digest,
842
+ candidateContentHash: experiment.candidate.digest
843
+ },
844
+ diff,
845
+ evaluation: {
846
+ generationsExplored: options.generationsExplored ?? 0,
847
+ searchDurationMs,
848
+ executionDurationMs: evaluation.executionDurationMs,
849
+ durationMs,
850
+ searchCostUsd,
851
+ executionCostUsd: evaluation.executionCostUsd,
852
+ totalCostUsd
853
+ },
854
+ ...options.metadata ? { metadata: options.metadata } : {}
855
+ });
856
+ const { recordDigest: _recordDigest, ...provenance } = provisional.provenance;
857
+ return agentImprovementMeasuredComparisonSchema.parse({
858
+ ...provisional,
859
+ provenance: {
860
+ ...provenance,
861
+ recordDigest: canonicalCandidateDigest({
862
+ ...provisional,
863
+ provenance
864
+ })
865
+ }
866
+ });
867
+ }
868
+ /** Recompute every statistic and decision from the signed experiment receipts. */
1054
869
  function verifyCandidateExperimentComparison(input) {
1055
- const comparison = agentImprovementMeasuredComparisonSchema.parse(input);
1056
- const recomputed = measuredComparisonFromCandidateExperiment({
1057
- experiment: comparison.experiment,
1058
- measurements: comparison.measurements,
1059
- runId: comparison.provenance.runId,
1060
- ...comparison.candidate ? { candidate: comparison.candidate } : {},
1061
- generationsExplored: comparison.evaluation.generationsExplored,
1062
- searchDurationMs: comparison.evaluation.searchDurationMs,
1063
- searchCostUsd: comparison.evaluation.searchCostUsd,
1064
- ...comparison.metadata ? { metadata: comparison.metadata } : {}
1065
- });
1066
- if (canonicalCandidateDigest(recomputed) !== canonicalCandidateDigest(comparison)) {
1067
- throw new Error("candidate experiment comparison does not match its Runtime receipts");
1068
- }
1069
- return comparison;
870
+ const comparison = agentImprovementMeasuredComparisonSchema.parse(input);
871
+ if (canonicalCandidateDigest(measuredComparisonFromCandidateExperiment({
872
+ experiment: comparison.experiment,
873
+ measurements: comparison.measurements,
874
+ runId: comparison.provenance.runId,
875
+ ...comparison.candidate ? { candidate: comparison.candidate } : {},
876
+ generationsExplored: comparison.evaluation.generationsExplored,
877
+ searchDurationMs: comparison.evaluation.searchDurationMs,
878
+ searchCostUsd: comparison.evaluation.searchCostUsd,
879
+ ...comparison.metadata ? { metadata: comparison.metadata } : {}
880
+ })) !== canonicalCandidateDigest(comparison)) throw new Error("candidate experiment comparison does not match its Runtime receipts");
881
+ return comparison;
1070
882
  }
1071
883
  function deriveCandidateBundleDiff(experiment) {
1072
- const surfaces = ["profile", "code", "execution", "knowledge", "memory"];
1073
- const changed = surfaces.flatMap((surface) => {
1074
- const baseline = experiment.baseline[surface] ?? null;
1075
- const candidate = experiment.candidate[surface] ?? null;
1076
- const baselineDigest = canonicalCandidateDigest(baseline);
1077
- const candidateDigest = canonicalCandidateDigest(candidate);
1078
- if (baselineDigest === candidateDigest) return [];
1079
- return [
1080
- [
1081
- `--- baseline/${surface} (${baselineDigest})`,
1082
- `+++ candidate/${surface} (${candidateDigest})`,
1083
- JSON.stringify({ baseline, candidate }, null, 2)
1084
- ].join("\n")
1085
- ];
1086
- });
1087
- if (changed.length === 0) {
1088
- throw new Error("candidate experiment has no changed candidate surface");
1089
- }
1090
- return changed.join("\n\n");
884
+ const changed = [
885
+ "profile",
886
+ "code",
887
+ "execution",
888
+ "knowledge",
889
+ "memory"
890
+ ].flatMap((surface) => {
891
+ const baseline = experiment.baseline[surface] ?? null;
892
+ const candidate = experiment.candidate[surface] ?? null;
893
+ const baselineDigest = canonicalCandidateDigest(baseline);
894
+ const candidateDigest = canonicalCandidateDigest(candidate);
895
+ if (baselineDigest === candidateDigest) return [];
896
+ return [[
897
+ `--- baseline/${surface} (${baselineDigest})`,
898
+ `+++ candidate/${surface} (${candidateDigest})`,
899
+ JSON.stringify({
900
+ baseline,
901
+ candidate
902
+ }, null, 2)
903
+ ].join("\n")];
904
+ });
905
+ if (changed.length === 0) throw new Error("candidate experiment has no changed candidate surface");
906
+ return changed.join("\n\n");
1091
907
  }
1092
908
  function verifyCandidateBenchmarkTask(input) {
1093
- const task = agentCandidateBenchmarkTaskSchema.parse(input);
1094
- verifySelfAddressed(task, "candidate benchmark task");
1095
- return task;
909
+ const task = agentCandidateBenchmarkTaskSchema.parse(input);
910
+ verifySelfAddressed(task, "candidate benchmark task");
911
+ return task;
1096
912
  }
1097
913
  function verifyCandidateBenchmarkSuiteInputs(input) {
1098
- if (input === null || typeof input !== "object" || Array.isArray(input)) {
1099
- throw new Error("candidate benchmark suite inputs must be an object");
1100
- }
1101
- const candidate = input;
1102
- const suite = verifyCandidateBenchmarkSuite(candidate.suite);
1103
- if (!Array.isArray(candidate.tasks) || candidate.tasks.length !== suite.taskDigests.length) {
1104
- throw new Error("candidate benchmark suite task count does not match its signed digests");
1105
- }
1106
- candidate.tasks.forEach((task, index) => {
1107
- const verified = verifyCandidateBenchmarkTask(task);
1108
- if (verified.digest !== suite.taskDigests[index]) {
1109
- throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
1110
- }
1111
- });
1112
- return { suite, tasks: candidate.tasks };
914
+ if (input === null || typeof input !== "object" || Array.isArray(input)) throw new Error("candidate benchmark suite inputs must be an object");
915
+ const candidate = input;
916
+ const suite = verifyCandidateBenchmarkSuite(candidate.suite);
917
+ if (!Array.isArray(candidate.tasks) || candidate.tasks.length !== suite.taskDigests.length) throw new Error("candidate benchmark suite task count does not match its signed digests");
918
+ candidate.tasks.forEach((task, index) => {
919
+ if (verifyCandidateBenchmarkTask(task).digest !== suite.taskDigests[index]) throw new Error(`candidate benchmark task ${index} does not match the signed suite`);
920
+ });
921
+ return {
922
+ suite,
923
+ tasks: candidate.tasks
924
+ };
1113
925
  }
1114
926
  function verifyCandidateBenchmarkSuite(input) {
1115
- const suite = agentCandidateBenchmarkSuiteSchema.parse(input);
1116
- verifySelfAddressed(suite, "candidate benchmark suite");
1117
- return suite;
927
+ const suite = agentCandidateBenchmarkSuiteSchema.parse(input);
928
+ verifySelfAddressed(suite, "candidate benchmark suite");
929
+ return suite;
1118
930
  }
1119
931
  function verifyBundle(input, label) {
1120
- const bundle = agentCandidateBundleSchema.parse(input);
1121
- verifySelfAddressed(bundle, label);
1122
- return bundle;
932
+ const bundle = agentCandidateBundleSchema.parse(input);
933
+ verifySelfAddressed(bundle, label);
934
+ return bundle;
1123
935
  }
1124
936
  function verifyMeasurement(experiment, input, index) {
1125
- const suite = experiment.benchmark.suite;
1126
- const taskIndex = Math.floor(index / suite.reps);
1127
- const repetition = index % suite.reps;
1128
- const task = experiment.benchmark.tasks[taskIndex];
1129
- const seed = suite.seeds[index];
1130
- if (!task || seed === void 0) {
1131
- throw new Error(`candidate experiment measurement ${index} is outside the signed suite`);
1132
- }
1133
- const baseline = verifyExecutionEvidence(input.baseline);
1134
- const candidate = verifyExecutionEvidence(input.candidate);
1135
- for (const [arm, evidence] of [
1136
- ["baseline", baseline],
1137
- ["candidate", candidate]
1138
- ]) {
1139
- const bundle = experiment[arm];
1140
- const materialization = evidence.materializationReceipt;
1141
- const plan = materialization.executionPlan.material;
1142
- const runCell = plan.runCell;
1143
- verifySelfAddressed(runCell, "candidate run cell");
1144
- if (runCell.experimentDigest !== experiment.digest || runCell.arm !== arm || runCell.bundleDigest !== bundle.digest || runCell.suiteDigest !== suite.digest || runCell.taskDigest !== task.digest || runCell.taskIndex !== taskIndex || runCell.repetition !== repetition || runCell.seed !== seed || runCell.attempt > task.attempt.maxAttempts || materialization.bundleDigest !== bundle.digest || materialization.benchmark.suite.digest !== suite.digest || materialization.benchmark.task.digest !== task.digest || materialization.codeKind !== bundle.code.kind || materialization.profileActivation.profilePlan.material.sourceProfileDigest !== canonicalCandidateDigest(bundle.profile) || evidence.receipt.runCellDigest !== runCell.digest || JSON.stringify(materialization.resolvedModel) !== JSON.stringify(task.model)) {
1145
- throw new Error(`candidate experiment measurement ${index} substituted its ${arm} arm`);
1146
- }
1147
- verifyTaskOutcome(task, evidence, index, arm);
1148
- }
1149
- const baselinePlan = baseline.materializationReceipt.executionPlan.material;
1150
- const candidatePlan = candidate.materializationReceipt.executionPlan.material;
1151
- if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) {
1152
- throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
1153
- }
1154
- return { baseline, candidate };
937
+ const suite = experiment.benchmark.suite;
938
+ const taskIndex = Math.floor(index / suite.reps);
939
+ const repetition = index % suite.reps;
940
+ const task = experiment.benchmark.tasks[taskIndex];
941
+ const seed = suite.seeds[index];
942
+ if (!task || seed === void 0) throw new Error(`candidate experiment measurement ${index} is outside the signed suite`);
943
+ const baseline = verifyExecutionEvidence(input.baseline);
944
+ const candidate = verifyExecutionEvidence(input.candidate);
945
+ for (const [arm, evidence] of [["baseline", baseline], ["candidate", candidate]]) {
946
+ const bundle = experiment[arm];
947
+ const materialization = evidence.materializationReceipt;
948
+ const runCell = materialization.executionPlan.material.runCell;
949
+ verifySelfAddressed(runCell, "candidate run cell");
950
+ if (runCell.experimentDigest !== experiment.digest || runCell.arm !== arm || runCell.bundleDigest !== bundle.digest || runCell.suiteDigest !== suite.digest || runCell.taskDigest !== task.digest || runCell.taskIndex !== taskIndex || runCell.repetition !== repetition || runCell.seed !== seed || runCell.attempt > task.attempt.maxAttempts || materialization.bundleDigest !== bundle.digest || materialization.benchmark.suite.digest !== suite.digest || materialization.benchmark.task.digest !== task.digest || materialization.codeKind !== bundle.code.kind || materialization.profileActivation.profilePlan.material.sourceProfileDigest !== canonicalCandidateDigest(bundle.profile) || evidence.receipt.runCellDigest !== runCell.digest || JSON.stringify(materialization.resolvedModel) !== JSON.stringify(task.model)) throw new Error(`candidate experiment measurement ${index} substituted its ${arm} arm`);
951
+ verifyTaskOutcome(task, evidence, index, arm);
952
+ }
953
+ const baselinePlan = baseline.materializationReceipt.executionPlan.material;
954
+ const candidatePlan = candidate.materializationReceipt.executionPlan.material;
955
+ if (baselinePlan.executionId === candidatePlan.executionId || baselinePlan.runCell.digest === candidatePlan.runCell.digest || baseline.materializationReceipt.digest === candidate.materializationReceipt.digest || baseline.receipt.digest === candidate.receipt.digest || baseline.digest === candidate.digest) throw new Error(`candidate experiment measurement ${index} reused one execution across arms`);
956
+ return {
957
+ baseline,
958
+ candidate
959
+ };
1155
960
  }
1156
961
  function verifyExecutionEvidence(input) {
1157
- const evidence = candidateExecutionEvidenceSchema.parse(input);
1158
- verifySelfAddressed(evidence, "candidate execution evidence");
1159
- verifySelfAddressed(evidence.materializationReceipt, "candidate materialization receipt");
1160
- verifySelfAddressed(
1161
- evidence.materializationReceipt.profileActivation,
1162
- "candidate profile activation"
1163
- );
1164
- verifyMaterialAddressed(
1165
- evidence.materializationReceipt.profileActivation.profilePlan,
1166
- "candidate profile plan"
1167
- );
1168
- verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
1169
- verifySelfAddressed(evidence.receipt, "candidate run receipt");
1170
- verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
1171
- verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
1172
- verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
1173
- return evidence;
962
+ const evidence = candidateExecutionEvidenceSchema.parse(input);
963
+ verifySelfAddressed(evidence, "candidate execution evidence");
964
+ verifySelfAddressed(evidence.materializationReceipt, "candidate materialization receipt");
965
+ verifySelfAddressed(evidence.materializationReceipt.profileActivation, "candidate profile activation");
966
+ verifyMaterialAddressed(evidence.materializationReceipt.profileActivation.profilePlan, "candidate profile plan");
967
+ verifyMaterialAddressed(evidence.materializationReceipt.executionPlan, "candidate execution plan");
968
+ verifySelfAddressed(evidence.receipt, "candidate run receipt");
969
+ verifyMaterialAddressed(evidence.receipt.modelSettlement, "candidate model settlement");
970
+ verifyMaterialAddressed(evidence.receipt.taskOutcome, "candidate task outcome");
971
+ verifyMaterialAddressed(evidence.receipt.benchmarkResult, "candidate benchmark result");
972
+ return evidence;
1174
973
  }
1175
974
  function verifySelfAddressed(document, label) {
1176
- if (canonicalCandidateDigest(omitTopLevelDigest(document)) !== document.digest) {
1177
- throw new Error(`${label} digest is invalid`);
1178
- }
975
+ if (canonicalCandidateDigest(omitTopLevelDigest(document)) !== document.digest) throw new Error(`${label} digest is invalid`);
1179
976
  }
1180
977
  function verifyTaskOutcome(task, evidence, index, arm) {
1181
- const outcome = evidence.receipt.taskOutcome.material.outcome;
1182
- const result = evidence.receipt.benchmarkResult.material;
1183
- const prefix = `candidate experiment measurement ${index} ${arm}`;
1184
- if (result.evidence.sha256 === task.grader.artifact.sha256) {
1185
- throw new Error(`${prefix} reused grader bytes as grading evidence`);
1186
- }
1187
- const usage = combinedUsage(evidence);
1188
- const usageChecks = [
1189
- [usage.modelCalls, task.limits.maxModelCalls, "model calls"],
1190
- [usage.inputTokens, task.limits.maxInputTokens, "input tokens"],
1191
- [usage.outputTokens, task.limits.maxOutputTokens, "output tokens"],
1192
- [usage.costUsdNanos, Math.round(task.limits.maxCostUsd * 1e9), "cost"]
1193
- ];
1194
- for (const [actual, maximum, label] of usageChecks) {
1195
- if (actual > maximum) {
1196
- throw new Error(`${prefix} ${label} ${actual} exceeds the signed limit ${maximum}`);
1197
- }
1198
- }
1199
- if (outcome.kind !== task.outcome.kind) {
1200
- throw new Error(`${prefix} returned an outcome outside the signed task contract`);
1201
- }
1202
- if (task.outcome.kind === "output") {
1203
- if (outcome.kind !== "output" || outcome.spec.mediaType !== task.outcome.mediaType || outcome.spec.maxBytes !== task.outcome.maxBytes) {
1204
- throw new Error(`${prefix} changed the signed output contract`);
1205
- }
1206
- return;
1207
- }
1208
- const repository = task.repository;
1209
- if (outcome.kind !== "workspace" || repository === void 0 || outcome.baseRepository.identity !== repository.identity || outcome.baseRepository.rootIdentity !== repository.rootIdentity || outcome.baseRepository.commit !== repository.baseCommit || outcome.baseRepository.tree !== repository.baseTree) {
1210
- throw new Error(`${prefix} did not start from the signed repository state`);
1211
- }
978
+ const outcome = evidence.receipt.taskOutcome.material.outcome;
979
+ const result = evidence.receipt.benchmarkResult.material;
980
+ const prefix = `candidate experiment measurement ${index} ${arm}`;
981
+ if (result.evidence.sha256 === task.grader.artifact.sha256) throw new Error(`${prefix} reused grader bytes as grading evidence`);
982
+ const usage = combinedUsage(evidence);
983
+ const usageChecks = [
984
+ [
985
+ usage.modelCalls,
986
+ task.limits.maxModelCalls,
987
+ "model calls"
988
+ ],
989
+ [
990
+ usage.inputTokens,
991
+ task.limits.maxInputTokens,
992
+ "input tokens"
993
+ ],
994
+ [
995
+ usage.outputTokens,
996
+ task.limits.maxOutputTokens,
997
+ "output tokens"
998
+ ],
999
+ [
1000
+ usage.costUsdNanos,
1001
+ Math.round(task.limits.maxCostUsd * 1e9),
1002
+ "cost"
1003
+ ]
1004
+ ];
1005
+ for (const [actual, maximum, label] of usageChecks) if (actual > maximum) throw new Error(`${prefix} ${label} ${actual} exceeds the signed limit ${maximum}`);
1006
+ if (outcome.kind !== task.outcome.kind) throw new Error(`${prefix} returned an outcome outside the signed task contract`);
1007
+ if (task.outcome.kind === "output") {
1008
+ if (outcome.kind !== "output" || outcome.spec.mediaType !== task.outcome.mediaType || outcome.spec.maxBytes !== task.outcome.maxBytes) throw new Error(`${prefix} changed the signed output contract`);
1009
+ return;
1010
+ }
1011
+ const repository = task.repository;
1012
+ if (outcome.kind !== "workspace" || repository === void 0 || outcome.baseRepository.identity !== repository.identity || outcome.baseRepository.rootIdentity !== repository.rootIdentity || outcome.baseRepository.commit !== repository.baseCommit || outcome.baseRepository.tree !== repository.baseTree) throw new Error(`${prefix} did not start from the signed repository state`);
1212
1013
  }
1213
1014
  function verifyStableProfileMaterialization(measurements) {
1214
- for (const arm of ["baseline", "candidate"]) {
1215
- const expected = measurements[0]?.[arm].materializationReceipt.profileActivation;
1216
- if (!expected) throw new Error("candidate experiment contains no profile materialization");
1217
- const expectedDigest = canonicalCandidateDigest({
1218
- profilePlanDigest: expected.profilePlan.digest,
1219
- files: expected.files
1220
- });
1221
- for (const [index, measurement] of measurements.entries()) {
1222
- const activation = measurement[arm].materializationReceipt.profileActivation;
1223
- if (canonicalCandidateDigest({
1224
- profilePlanDigest: activation.profilePlan.digest,
1225
- files: activation.files
1226
- }) !== expectedDigest) {
1227
- throw new Error(
1228
- `candidate experiment measurement ${index} ${arm} materialized a different profile`
1229
- );
1230
- }
1231
- }
1232
- }
1015
+ for (const arm of ["baseline", "candidate"]) {
1016
+ const expected = measurements[0]?.[arm].materializationReceipt.profileActivation;
1017
+ if (!expected) throw new Error("candidate experiment contains no profile materialization");
1018
+ const expectedDigest = canonicalCandidateDigest({
1019
+ profilePlanDigest: expected.profilePlan.digest,
1020
+ files: expected.files
1021
+ });
1022
+ for (const [index, measurement] of measurements.entries()) {
1023
+ const activation = measurement[arm].materializationReceipt.profileActivation;
1024
+ if (canonicalCandidateDigest({
1025
+ profilePlanDigest: activation.profilePlan.digest,
1026
+ files: activation.files
1027
+ }) !== expectedDigest) throw new Error(`candidate experiment measurement ${index} ${arm} materialized a different profile`);
1028
+ }
1029
+ }
1233
1030
  }
1234
1031
  function completedSuccessfully(evidence) {
1235
- const termination = evidence.receipt.termination;
1236
- return termination.kind === "exit" && termination.exitCode === 0;
1032
+ const termination = evidence.receipt.termination;
1033
+ return termination.kind === "exit" && termination.exitCode === 0;
1237
1034
  }
1238
1035
  function verifyMaterialAddressed(evidence, label) {
1239
- if (canonicalCandidateDigest(evidence.material) !== evidence.digest) {
1240
- throw new Error(`${label} digest is invalid`);
1241
- }
1036
+ if (canonicalCandidateDigest(evidence.material) !== evidence.digest) throw new Error(`${label} digest is invalid`);
1242
1037
  }
1243
1038
  function projectPairedMeasurement(measurement, index, adapter) {
1244
- if (typeof measurement.cellId !== "string" || !measurement.cellId.trim()) {
1245
- throw new Error(`paired measurement ${index} requires a cell id`);
1246
- }
1247
- return {
1248
- cellId: measurement.cellId,
1249
- baseline: projectRun(measurement.baseline, adapter, `paired measurement ${index} baseline`),
1250
- candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
1251
- };
1039
+ if (typeof measurement.cellId !== "string" || !measurement.cellId.trim()) throw new Error(`paired measurement ${index} requires a cell id`);
1040
+ return {
1041
+ cellId: measurement.cellId,
1042
+ baseline: projectRun(measurement.baseline, adapter, `paired measurement ${index} baseline`),
1043
+ candidate: projectRun(measurement.candidate, adapter, `paired measurement ${index} candidate`)
1044
+ };
1252
1045
  }
1253
1046
  function projectRun(run, adapter, label) {
1254
- const suppliedDimensions = adapter.dimensions(run);
1255
- if (!Array.isArray(suppliedDimensions)) {
1256
- throw new Error(`${label} dimensions must be an array`);
1257
- }
1258
- const dimensions = /* @__PURE__ */ new Map();
1259
- for (const dimension of suppliedDimensions) {
1260
- if (typeof dimension.name !== "string" || !dimension.name.trim()) {
1261
- throw new Error(`${label} contains an unnamed dimension`);
1262
- }
1263
- if (dimensions.has(dimension.name)) {
1264
- throw new Error(`${label} repeats dimension '${dimension.name}'`);
1265
- }
1266
- dimensions.set(dimension.name, finiteMeasurement(dimension.score, `${label} ${dimension.name}`));
1267
- }
1268
- const completed = adapter.completed(run);
1269
- const passed = adapter.passed(run);
1270
- if (typeof completed !== "boolean" || typeof passed !== "boolean") {
1271
- throw new Error(`${label} completion and pass values must be booleans`);
1272
- }
1273
- return {
1274
- score: finiteMeasurement(adapter.score(run), `${label} score`),
1275
- dimensions,
1276
- costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
1277
- latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
1278
- completed,
1279
- passed
1280
- };
1047
+ const suppliedDimensions = adapter.dimensions(run);
1048
+ if (!Array.isArray(suppliedDimensions)) throw new Error(`${label} dimensions must be an array`);
1049
+ const dimensions = /* @__PURE__ */ new Map();
1050
+ for (const dimension of suppliedDimensions) {
1051
+ if (typeof dimension.name !== "string" || !dimension.name.trim()) throw new Error(`${label} contains an unnamed dimension`);
1052
+ if (dimensions.has(dimension.name)) throw new Error(`${label} repeats dimension '${dimension.name}'`);
1053
+ dimensions.set(dimension.name, finiteMeasurement(dimension.score, `${label} ${dimension.name}`));
1054
+ }
1055
+ const completed = adapter.completed(run);
1056
+ const passed = adapter.passed(run);
1057
+ if (typeof completed !== "boolean" || typeof passed !== "boolean") throw new Error(`${label} completion and pass values must be booleans`);
1058
+ return {
1059
+ score: finiteMeasurement(adapter.score(run), `${label} score`),
1060
+ dimensions,
1061
+ costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
1062
+ latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
1063
+ completed,
1064
+ passed
1065
+ };
1281
1066
  }
1282
1067
  function sharedProjectedDimensions(measurements) {
1283
- const expected = [...measurements[0].baseline.dimensions.keys()];
1284
- for (const [index, measurement] of measurements.entries()) {
1285
- for (const [arm, run] of [
1286
- ["baseline", measurement.baseline],
1287
- ["candidate", measurement.candidate]
1288
- ]) {
1289
- const actual = [...run.dimensions.keys()];
1290
- if (JSON.stringify(actual) !== JSON.stringify(expected)) {
1291
- throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
1292
- }
1293
- }
1294
- }
1295
- return expected;
1068
+ const expected = [...measurements[0].baseline.dimensions.keys()];
1069
+ for (const [index, measurement] of measurements.entries()) for (const [arm, run] of [["baseline", measurement.baseline], ["candidate", measurement.candidate]]) {
1070
+ const actual = [...run.dimensions.keys()];
1071
+ if (JSON.stringify(actual) !== JSON.stringify(expected)) throw new Error(`paired measurement ${index} ${arm} dimensions do not match the suite`);
1072
+ }
1073
+ return expected;
1296
1074
  }
1297
1075
  function dimensionScore(run, name) {
1298
- const value = run.dimensions.get(name);
1299
- if (value === void 0) throw new Error(`paired measurement is missing dimension '${name}'`);
1300
- return value;
1076
+ const value = run.dimensions.get(name);
1077
+ if (value === void 0) throw new Error(`paired measurement is missing dimension '${name}'`);
1078
+ return value;
1301
1079
  }
1302
1080
  function measuredEstimate(baseline, candidate, options) {
1303
- const bootstrap = pairedBootstrap(baseline, candidate, {
1304
- confidence: options.confidence,
1305
- resamples: options.resamples,
1306
- statistic: "mean",
1307
- seed: options.seed
1308
- });
1309
- const baselineMean = mean(baseline);
1310
- const candidateMean = mean(candidate);
1311
- const delta = candidateMean - baselineMean;
1312
- return {
1313
- baseline: baselineMean,
1314
- candidate: candidateMean,
1315
- delta,
1316
- confidenceInterval: {
1317
- level: bootstrap.confidence,
1318
- lower: Math.min(bootstrap.low, delta),
1319
- upper: Math.max(bootstrap.high, delta),
1320
- method: "paired-bootstrap",
1321
- statistic: "mean",
1322
- resamples: bootstrap.resamples
1323
- },
1324
- n: bootstrap.n
1325
- };
1081
+ const bootstrap = pairedBootstrap(baseline, candidate, {
1082
+ confidence: options.confidence,
1083
+ resamples: options.resamples,
1084
+ statistic: "mean",
1085
+ seed: options.seed
1086
+ });
1087
+ const baselineMean = mean(baseline);
1088
+ const candidateMean = mean(candidate);
1089
+ const delta = candidateMean - baselineMean;
1090
+ return {
1091
+ baseline: baselineMean,
1092
+ candidate: candidateMean,
1093
+ delta,
1094
+ confidenceInterval: {
1095
+ level: bootstrap.confidence,
1096
+ lower: Math.min(bootstrap.low, delta),
1097
+ upper: Math.max(bootstrap.high, delta),
1098
+ method: "paired-bootstrap",
1099
+ statistic: "mean",
1100
+ resamples: bootstrap.resamples
1101
+ },
1102
+ n: bootstrap.n
1103
+ };
1326
1104
  }
1327
1105
  function finiteMeasurement(value, label) {
1328
- if (!Number.isFinite(value)) throw new Error(`${label} must be finite`);
1329
- return value;
1106
+ if (!Number.isFinite(value)) throw new Error(`${label} must be finite`);
1107
+ return value;
1330
1108
  }
1331
1109
  function nonNegativeMeasurement(value, label) {
1332
- if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be non-negative`);
1333
- return value;
1334
- }
1335
- var candidateExecutionEvidenceAdapter = {
1336
- score: (evidence) => evidence.receipt.benchmarkResult.material.score,
1337
- dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
1338
- costUsd: costFromEvidence,
1339
- latencyMs: latencyFromEvidence,
1340
- completed: completedSuccessfully,
1341
- passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
1110
+ if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be non-negative`);
1111
+ return value;
1112
+ }
1113
+ const candidateExecutionEvidenceAdapter = {
1114
+ score: (evidence) => evidence.receipt.benchmarkResult.material.score,
1115
+ dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
1116
+ costUsd: costFromEvidence,
1117
+ latencyMs: latencyFromEvidence,
1118
+ completed: completedSuccessfully,
1119
+ passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
1342
1120
  };
1343
1121
  function costFromEvidence(evidence) {
1344
- return combinedUsage(evidence).costUsdNanos / 1e9;
1122
+ return combinedUsage(evidence).costUsdNanos / 1e9;
1345
1123
  }
1346
1124
  function latencyFromEvidence(evidence) {
1347
- return evidence.receipt.timing.durationMs + evidence.receipt.benchmarkResult.material.grading.timing.durationMs;
1125
+ return evidence.receipt.timing.durationMs + evidence.receipt.benchmarkResult.material.grading.timing.durationMs;
1348
1126
  }
1349
1127
  function combinedUsage(evidence) {
1350
- const candidate = evidence.receipt.modelSettlement.material.usage;
1351
- const grader = evidence.receipt.benchmarkResult.material.grading.usage;
1352
- return {
1353
- inputTokens: candidate.inputTokens + grader.inputTokens,
1354
- outputTokens: candidate.outputTokens + grader.outputTokens,
1355
- cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
1356
- reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
1357
- modelCalls: candidate.modelCalls + grader.modelCalls,
1358
- costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
1359
- };
1128
+ const candidate = evidence.receipt.modelSettlement.material.usage;
1129
+ const grader = evidence.receipt.benchmarkResult.material.grading.usage;
1130
+ return {
1131
+ inputTokens: candidate.inputTokens + grader.inputTokens,
1132
+ outputTokens: candidate.outputTokens + grader.outputTokens,
1133
+ cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
1134
+ reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
1135
+ modelCalls: candidate.modelCalls + grader.modelCalls,
1136
+ costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
1137
+ };
1360
1138
  }
1361
1139
  function cellIds(experiment) {
1362
- const { suite, tasks } = experiment.benchmark;
1363
- return suite.seeds.map((_, index) => {
1364
- const taskIndex = Math.floor(index / suite.reps);
1365
- const repetition = index % suite.reps;
1366
- return `${tasks[taskIndex]?.scenario.id ?? taskIndex}:${repetition}`;
1367
- });
1140
+ const { suite, tasks } = experiment.benchmark;
1141
+ return suite.seeds.map((_, index) => {
1142
+ const taskIndex = Math.floor(index / suite.reps);
1143
+ const repetition = index % suite.reps;
1144
+ return `${tasks[taskIndex]?.scenario.id ?? taskIndex}:${repetition}`;
1145
+ });
1368
1146
  }
1369
1147
  function mean(values) {
1370
- if (values.length === 0) throw new Error("candidate experiment requires measured values");
1371
- return values.reduce((sum, value) => sum + value, 0) / values.length;
1148
+ if (values.length === 0) throw new Error("candidate experiment requires measured values");
1149
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
1372
1150
  }
1373
1151
  function abortError(signal) {
1374
- return signal.reason instanceof Error ? signal.reason : new Error("candidate experiment aborted");
1375
- }
1376
-
1377
- // src/contract/eval-reporting-suite.ts
1378
- import { mkdir, writeFile } from "fs/promises";
1379
- import { dirname, join as join2 } from "path";
1380
-
1381
- // src/contract/intake/run-record-dir.ts
1382
- import { readdir, readFile, stat } from "fs/promises";
1383
- import { join } from "path";
1384
- var ANALYSIS_ARTIFACT = "analysis.json";
1152
+ return signal.reason instanceof Error ? signal.reason : /* @__PURE__ */ new Error("candidate experiment aborted");
1153
+ }
1154
+ //#endregion
1155
+ //#region src/contract/intake/run-record-dir.ts
1156
+ /**
1157
+ * # `intake/run-record-dir` load a directory or file of `RunRecord`s.
1158
+ *
1159
+ * The on-disk counterpart to the in-memory intake adapters: point it at a
1160
+ * single `.json` (array) / `.jsonl` (one record per line) file or at a
1161
+ * directory of such files, and it returns the substrate-canonical
1162
+ * `RunRecord[]` ready for `analyzeRuns({ runs })`.
1163
+ *
1164
+ * Validation is at the boundary: each parsed object goes through
1165
+ * `parseRunRecordSafe`. By default an invalid record fails loud with its
1166
+ * file + index; pass `onInvalid: 'collect'` to keep the valid records and
1167
+ * receive the rejects as structured diagnostics instead.
1168
+ */
1169
+ const ANALYSIS_ARTIFACT$1 = "analysis.json";
1385
1170
  function defaultInclude(fileName) {
1386
- if (fileName === ANALYSIS_ARTIFACT) return false;
1387
- return fileName.endsWith(".json") || fileName.endsWith(".jsonl");
1388
- }
1171
+ if (fileName === ANALYSIS_ARTIFACT$1) return false;
1172
+ return fileName.endsWith(".json") || fileName.endsWith(".jsonl");
1173
+ }
1174
+ /**
1175
+ * Resolve a file or directory path into validated `RunRecord[]`.
1176
+ *
1177
+ * A `.json` file must parse to a top-level array; a `.jsonl` file is one
1178
+ * record per non-empty line. Directories are read shallowly by default
1179
+ * (set `recursive` to descend); the `analysis.json` output artifact is
1180
+ * always excluded.
1181
+ */
1389
1182
  async function fromRunRecordDir(path, options = {}) {
1390
- const onInvalid = options.onInvalid ?? "throw";
1391
- const include = options.include ?? defaultInclude;
1392
- const stats = await stat(path);
1393
- const filePaths = stats.isDirectory() ? await collectFiles(path, include, options.recursive ?? false) : [path];
1394
- const runs = [];
1395
- const rejected = [];
1396
- for (const file of filePaths) {
1397
- const raw = await parseRecordFile(file);
1398
- for (const { index, value } of raw) {
1399
- const parsed = parseRunRecordSafe(value);
1400
- if (parsed.ok) {
1401
- runs.push(parsed.value);
1402
- continue;
1403
- }
1404
- const rejection = { file, index, reason: parsed.error.message };
1405
- if (onInvalid === "throw") {
1406
- throw new Error(
1407
- `fromRunRecordDir: invalid RunRecord in '${file}' at index ${index}: ${parsed.error.message}`
1408
- );
1409
- }
1410
- rejected.push(rejection);
1411
- }
1412
- }
1413
- return { runs, rejected, files: filePaths };
1414
- }
1183
+ const onInvalid = options.onInvalid ?? "throw";
1184
+ const include = options.include ?? defaultInclude;
1185
+ const filePaths = (await stat(path)).isDirectory() ? await collectFiles(path, include, options.recursive ?? false) : [path];
1186
+ const runs = [];
1187
+ const rejected = [];
1188
+ for (const file of filePaths) {
1189
+ const raw = await parseRecordFile(file);
1190
+ for (const { index, value } of raw) {
1191
+ const parsed = parseRunRecordSafe(value);
1192
+ if (parsed.ok) {
1193
+ runs.push(parsed.value);
1194
+ continue;
1195
+ }
1196
+ const rejection = {
1197
+ file,
1198
+ index,
1199
+ reason: parsed.error.message
1200
+ };
1201
+ if (onInvalid === "throw") throw new Error(`fromRunRecordDir: invalid RunRecord in '${file}' at index ${index}: ${parsed.error.message}`);
1202
+ rejected.push(rejection);
1203
+ }
1204
+ }
1205
+ return {
1206
+ runs,
1207
+ rejected,
1208
+ files: filePaths
1209
+ };
1210
+ }
1211
+ /** Read a single `.json` / `.jsonl` file into `{ index, value }` pairs. A
1212
+ * malformed JSONL line throws with its line number rather than being skipped —
1213
+ * silent line-dropping is how corpora quietly shrink. */
1415
1214
  async function parseRecordFile(file) {
1416
- const text = await readFile(file, "utf8");
1417
- const trimmed = text.trim();
1418
- if (trimmed.length === 0) return [];
1419
- if (trimmed.startsWith("[")) {
1420
- const parsed = JSON.parse(trimmed);
1421
- if (!Array.isArray(parsed)) {
1422
- throw new Error(`fromRunRecordDir: file '${file}' did not parse to an array`);
1423
- }
1424
- return parsed.map((value, index) => ({ index, value }));
1425
- }
1426
- const out = [];
1427
- const lines = trimmed.split("\n");
1428
- for (let i = 0; i < lines.length; i++) {
1429
- const line = lines[i].trim();
1430
- if (line.length === 0) continue;
1431
- try {
1432
- out.push({ index: i, value: JSON.parse(line) });
1433
- } catch (err) {
1434
- throw new Error(
1435
- `fromRunRecordDir: file '${file}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`
1436
- );
1437
- }
1438
- }
1439
- return out;
1440
- }
1215
+ const trimmed = (await readFile(file, "utf8")).trim();
1216
+ if (trimmed.length === 0) return [];
1217
+ if (trimmed.startsWith("[")) {
1218
+ const parsed = JSON.parse(trimmed);
1219
+ if (!Array.isArray(parsed)) throw new Error(`fromRunRecordDir: file '${file}' did not parse to an array`);
1220
+ return parsed.map((value, index) => ({
1221
+ index,
1222
+ value
1223
+ }));
1224
+ }
1225
+ const out = [];
1226
+ const lines = trimmed.split("\n");
1227
+ for (let i = 0; i < lines.length; i++) {
1228
+ const line = lines[i].trim();
1229
+ if (line.length === 0) continue;
1230
+ try {
1231
+ out.push({
1232
+ index: i,
1233
+ value: JSON.parse(line)
1234
+ });
1235
+ } catch (err) {
1236
+ throw new Error(`fromRunRecordDir: file '${file}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`);
1237
+ }
1238
+ }
1239
+ return out;
1240
+ }
1241
+ /** Sorted file list under a directory, filtered by `include`. Sorted so the
1242
+ * resulting `RunRecord` order — and any downstream fingerprint — is stable
1243
+ * across filesystems. */
1441
1244
  async function collectFiles(dir, include, recursive) {
1442
- const entries = await readdir(dir, { withFileTypes: true });
1443
- const files = [];
1444
- const subdirs = [];
1445
- for (const entry of entries) {
1446
- if (entry.isDirectory()) {
1447
- if (recursive) subdirs.push(join(dir, entry.name));
1448
- continue;
1449
- }
1450
- if (include(entry.name)) files.push(join(dir, entry.name));
1451
- }
1452
- files.sort();
1453
- subdirs.sort();
1454
- for (const sub of subdirs) {
1455
- files.push(...await collectFiles(sub, include, recursive));
1456
- }
1457
- return files;
1458
- }
1459
-
1460
- // src/contract/eval-reporting-suite.ts
1461
- var ANALYSIS_ARTIFACT2 = "analysis.json";
1245
+ const entries = await readdir(dir, { withFileTypes: true });
1246
+ const files = [];
1247
+ const subdirs = [];
1248
+ for (const entry of entries) {
1249
+ if (entry.isDirectory()) {
1250
+ if (recursive) subdirs.push(join(dir, entry.name));
1251
+ continue;
1252
+ }
1253
+ if (include(entry.name)) files.push(join(dir, entry.name));
1254
+ }
1255
+ files.sort();
1256
+ subdirs.sort();
1257
+ for (const sub of subdirs) files.push(...await collectFiles(sub, include, recursive));
1258
+ return files;
1259
+ }
1260
+ //#endregion
1261
+ //#region src/contract/eval-reporting-suite.ts
1262
+ /**
1263
+ * # `evalReportingSuite` — one call from runs (or a run dir) to `analysis.json`.
1264
+ *
1265
+ * A thin wrapper over the analysis primitive (`analyzeRuns`) and the on-disk
1266
+ * intake adapter (`fromRunRecordDir`). It does NOT reimplement any statistics,
1267
+ * distributions, or clustering — it resolves the input into validated
1268
+ * `RunRecord[]`, calls `analyzeRuns` with the options you'd pass it directly,
1269
+ * wraps the result in a small provenance envelope, and (optionally) writes a
1270
+ * single `analysis.json` artifact.
1271
+ *
1272
+ * ```ts
1273
+ * // From a directory of run files, write ./runs/analysis.json:
1274
+ * const suite = await evalReportingSuite('./runs', { write: true })
1275
+ * // From records already in memory, no write:
1276
+ * const suite = await evalReportingSuite(records, { analyze: { decisionThreshold: 0.03 } })
1277
+ * suite.report // the InsightReport — distributions, paired lift, findings rollup
1278
+ * ```
1279
+ */
1280
+ const ANALYSIS_ARTIFACT = "analysis.json";
1281
+ /**
1282
+ * Resolve runs (or a run dir/file), run `analyzeRuns`, and optionally persist a
1283
+ * single `analysis.json`. The only analysis logic lives in `analyzeRuns`; this
1284
+ * function is composition + I/O.
1285
+ */
1462
1286
  async function evalReportingSuite(input, options = {}) {
1463
- const fromPath = typeof input === "string";
1464
- let runs;
1465
- let files = [];
1466
- let rejected = [];
1467
- if (fromPath) {
1468
- const loaded = await fromRunRecordDir(input, options.load);
1469
- runs = loaded.runs;
1470
- files = loaded.files;
1471
- rejected = loaded.rejected;
1472
- } else {
1473
- runs = input;
1474
- }
1475
- if (runs.length === 0) {
1476
- throw new Error(
1477
- fromPath ? `evalReportingSuite: no RunRecords found at '${input}'` : "evalReportingSuite: no RunRecords to analyze"
1478
- );
1479
- }
1480
- const report = await analyzeRuns({ ...options.analyze, runs });
1481
- const result = {
1482
- report,
1483
- provenance: {
1484
- generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
1485
- runCount: runs.length,
1486
- sourcePath: fromPath ? input : null,
1487
- files,
1488
- rejected
1489
- },
1490
- writtenTo: null
1491
- };
1492
- const target = resolveWriteTarget(options.write, fromPath ? input : null);
1493
- if (target) {
1494
- await mkdir(dirname(target), { recursive: true });
1495
- await writeFile(target, `${JSON.stringify(result, null, 2)}
1496
- `, "utf8");
1497
- result.writtenTo = target;
1498
- }
1499
- return result;
1500
- }
1287
+ const fromPath = typeof input === "string";
1288
+ let runs;
1289
+ let files = [];
1290
+ let rejected = [];
1291
+ if (fromPath) {
1292
+ const loaded = await fromRunRecordDir(input, options.load);
1293
+ runs = loaded.runs;
1294
+ files = loaded.files;
1295
+ rejected = loaded.rejected;
1296
+ } else runs = input;
1297
+ if (runs.length === 0) throw new Error(fromPath ? `evalReportingSuite: no RunRecords found at '${input}'` : "evalReportingSuite: no RunRecords to analyze");
1298
+ const result = {
1299
+ report: await analyzeRuns({
1300
+ ...options.analyze,
1301
+ runs
1302
+ }),
1303
+ provenance: {
1304
+ generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
1305
+ runCount: runs.length,
1306
+ sourcePath: fromPath ? input : null,
1307
+ files,
1308
+ rejected
1309
+ },
1310
+ writtenTo: null
1311
+ };
1312
+ const target = resolveWriteTarget(options.write, fromPath ? input : null);
1313
+ if (target) {
1314
+ await mkdir(dirname(target), { recursive: true });
1315
+ await writeFile(target, `${JSON.stringify(result, null, 2)}\n`, "utf8");
1316
+ result.writtenTo = target;
1317
+ }
1318
+ return result;
1319
+ }
1320
+ /** Resolve where (if anywhere) to write `analysis.json`. Returns null when
1321
+ * writing is disabled. Throws on `write: true` with in-memory input — there is
1322
+ * no directory to anchor the artifact to, and silently inventing `cwd` would
1323
+ * scatter files. */
1501
1324
  function resolveWriteTarget(write, sourcePath) {
1502
- if (!write) return null;
1503
- if (typeof write === "string") {
1504
- const looksLikeDir = write.endsWith("/") || !write.endsWith(".json") && !write.endsWith(".jsonl");
1505
- return looksLikeDir ? join2(write, ANALYSIS_ARTIFACT2) : write;
1506
- }
1507
- if (sourcePath === null) {
1508
- throw new Error(
1509
- "evalReportingSuite: write:true needs a source path to anchor analysis.json \u2014 pass an explicit output path when analyzing in-memory records"
1510
- );
1511
- }
1512
- const isFile = sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl");
1513
- return isFile ? join2(dirname(sourcePath), ANALYSIS_ARTIFACT2) : join2(sourcePath, ANALYSIS_ARTIFACT2);
1325
+ if (!write) return null;
1326
+ if (typeof write === "string") return write.endsWith("/") || !write.endsWith(".json") && !write.endsWith(".jsonl") ? join(write, ANALYSIS_ARTIFACT) : write;
1327
+ if (sourcePath === null) throw new Error("evalReportingSuite: write:true needs a source path to anchor analysis.json pass an explicit output path when analyzing in-memory records");
1328
+ return sourcePath.endsWith(".json") || sourcePath.endsWith(".jsonl") ? join(dirname(sourcePath), ANALYSIS_ARTIFACT) : join(sourcePath, ANALYSIS_ARTIFACT);
1514
1329
  }
1515
-
1516
- // src/contract/diff.ts
1330
+ //#endregion
1331
+ //#region src/contract/diff.ts
1517
1332
  function keyForCell(cell) {
1518
- return JSON.stringify([cell.scenarioId, cell.rep]);
1333
+ return JSON.stringify([cell.scenarioId, cell.rep]);
1519
1334
  }
1335
+ /** Build the per-dimension delta map for a matched cell. Each judge name +
1336
+ * dimension name encountered on EITHER side appears in the result. */
1520
1337
  function diffDimensions(before, after) {
1521
- const out = {};
1522
- const judges = /* @__PURE__ */ new Set([...Object.keys(before), ...Object.keys(after)]);
1523
- for (const judge of judges) {
1524
- const beforeDims = before[judge] ?? {};
1525
- const afterDims = after[judge] ?? {};
1526
- const dims = /* @__PURE__ */ new Set([...Object.keys(beforeDims), ...Object.keys(afterDims)]);
1527
- const judgeOut = {};
1528
- for (const dim of dims) {
1529
- const rawBefore = beforeDims[dim];
1530
- const rawAfter = afterDims[dim];
1531
- const b = typeof rawBefore === "number" && Number.isFinite(rawBefore) ? rawBefore : null;
1532
- const a = typeof rawAfter === "number" && Number.isFinite(rawAfter) ? rawAfter : null;
1533
- judgeOut[dim] = {
1534
- before: b,
1535
- after: a,
1536
- delta: b !== null && a !== null ? a - b : null
1537
- };
1538
- }
1539
- out[judge] = judgeOut;
1540
- }
1541
- return out;
1542
- }
1338
+ const out = {};
1339
+ const judges = /* @__PURE__ */ new Set([...Object.keys(before), ...Object.keys(after)]);
1340
+ for (const judge of judges) {
1341
+ const beforeDims = before[judge] ?? {};
1342
+ const afterDims = after[judge] ?? {};
1343
+ const dims = /* @__PURE__ */ new Set([...Object.keys(beforeDims), ...Object.keys(afterDims)]);
1344
+ const judgeOut = {};
1345
+ for (const dim of dims) {
1346
+ const rawBefore = beforeDims[dim];
1347
+ const rawAfter = afterDims[dim];
1348
+ const b = typeof rawBefore === "number" && Number.isFinite(rawBefore) ? rawBefore : null;
1349
+ const a = typeof rawAfter === "number" && Number.isFinite(rawAfter) ? rawAfter : null;
1350
+ judgeOut[dim] = {
1351
+ before: b,
1352
+ after: a,
1353
+ delta: b !== null && a !== null ? a - b : null
1354
+ };
1355
+ }
1356
+ out[judge] = judgeOut;
1357
+ }
1358
+ return out;
1359
+ }
1360
+ /**
1361
+ * Diff two generation snapshots. Cells are matched on `(scenarioId, rep)`;
1362
+ * unmatched cells surface in `added` / `removed`. Aggregate fields are
1363
+ * recomputed from the snapshot's stored fields, not re-derived from cells —
1364
+ * this keeps the diff consistent with whatever aggregation the substrate
1365
+ * actually reported.
1366
+ */
1543
1367
  function diffGenerations(before, after) {
1544
- const beforeMap = new Map(before.cells.map((c) => [keyForCell(c), c]));
1545
- const afterMap = new Map(after.cells.map((c) => [keyForCell(c), c]));
1546
- const matched = [];
1547
- const removed = [];
1548
- const added = [];
1549
- for (const [key, beforeCell] of beforeMap) {
1550
- const afterCell = afterMap.get(key);
1551
- if (!afterCell) {
1552
- removed.push(beforeCell);
1553
- continue;
1554
- }
1555
- matched.push({
1556
- scenarioId: beforeCell.scenarioId,
1557
- rep: beforeCell.rep,
1558
- compositeBefore: beforeCell.compositeMean,
1559
- compositeAfter: afterCell.compositeMean,
1560
- compositeDelta: beforeCell.compositeMean === null || afterCell.compositeMean === null ? null : afterCell.compositeMean - beforeCell.compositeMean,
1561
- dimensions: diffDimensions(beforeCell.dimensions, afterCell.dimensions)
1562
- });
1563
- }
1564
- for (const [key, afterCell] of afterMap) {
1565
- if (!beforeMap.has(key)) added.push(afterCell);
1566
- }
1567
- return {
1568
- beforeIndex: before.index,
1569
- afterIndex: after.index,
1570
- beforeSurfaceHash: before.surfaceHash,
1571
- afterSurfaceHash: after.surfaceHash,
1572
- surfaceChanged: before.surfaceHash !== after.surfaceHash,
1573
- matched,
1574
- removed,
1575
- added,
1576
- compositeBefore: before.compositeMean,
1577
- compositeAfter: after.compositeMean,
1578
- compositeDelta: before.compositeMean === null || after.compositeMean === null ? null : after.compositeMean - before.compositeMean,
1579
- costUsdBefore: before.costUsd,
1580
- costUsdAfter: after.costUsd,
1581
- costUsdDelta: after.costUsd - before.costUsd,
1582
- durationMsBefore: before.durationMs,
1583
- durationMsAfter: after.durationMs,
1584
- durationMsDelta: after.durationMs - before.durationMs
1585
- };
1586
- }
1368
+ const beforeMap = new Map(before.cells.map((c) => [keyForCell(c), c]));
1369
+ const afterMap = new Map(after.cells.map((c) => [keyForCell(c), c]));
1370
+ const matched = [];
1371
+ const removed = [];
1372
+ const added = [];
1373
+ for (const [key, beforeCell] of beforeMap) {
1374
+ const afterCell = afterMap.get(key);
1375
+ if (!afterCell) {
1376
+ removed.push(beforeCell);
1377
+ continue;
1378
+ }
1379
+ matched.push({
1380
+ scenarioId: beforeCell.scenarioId,
1381
+ rep: beforeCell.rep,
1382
+ compositeBefore: beforeCell.compositeMean,
1383
+ compositeAfter: afterCell.compositeMean,
1384
+ compositeDelta: beforeCell.compositeMean === null || afterCell.compositeMean === null ? null : afterCell.compositeMean - beforeCell.compositeMean,
1385
+ dimensions: diffDimensions(beforeCell.dimensions, afterCell.dimensions)
1386
+ });
1387
+ }
1388
+ for (const [key, afterCell] of afterMap) if (!beforeMap.has(key)) added.push(afterCell);
1389
+ return {
1390
+ beforeIndex: before.index,
1391
+ afterIndex: after.index,
1392
+ beforeSurfaceHash: before.surfaceHash,
1393
+ afterSurfaceHash: after.surfaceHash,
1394
+ surfaceChanged: before.surfaceHash !== after.surfaceHash,
1395
+ matched,
1396
+ removed,
1397
+ added,
1398
+ compositeBefore: before.compositeMean,
1399
+ compositeAfter: after.compositeMean,
1400
+ compositeDelta: before.compositeMean === null || after.compositeMean === null ? null : after.compositeMean - before.compositeMean,
1401
+ costUsdBefore: before.costUsd,
1402
+ costUsdAfter: after.costUsd,
1403
+ costUsdDelta: after.costUsd - before.costUsd,
1404
+ durationMsBefore: before.durationMs,
1405
+ durationMsAfter: after.durationMs,
1406
+ durationMsDelta: after.durationMs - before.durationMs
1407
+ };
1408
+ }
1409
+ /** Highest-index generation, or null if the run recorded none. */
1587
1410
  function winnerOf(run) {
1588
- if (run.generations.length === 0) return null;
1589
- let winner = run.generations[0];
1590
- for (const gen of run.generations) {
1591
- if (gen.index > winner.index) winner = gen;
1592
- }
1593
- return winner;
1594
- }
1411
+ if (run.generations.length === 0) return null;
1412
+ let winner = run.generations[0];
1413
+ for (const gen of run.generations) if (gen.index > winner.index) winner = gen;
1414
+ return winner;
1415
+ }
1416
+ /**
1417
+ * Diff two full eval-runs. Produces baseline-vs-baseline and
1418
+ * winner-vs-winner generation diffs when both sides expose them, plus
1419
+ * run-level cost / lift / gate-decision deltas.
1420
+ */
1595
1421
  function diffRuns(before, after) {
1596
- const beforeWinner = winnerOf(before);
1597
- const afterWinner = winnerOf(after);
1598
- const baselineDiff = before.baseline && after.baseline ? diffGenerations(before.baseline, after.baseline) : null;
1599
- const winnersDiff = beforeWinner && afterWinner ? diffGenerations(beforeWinner, afterWinner) : null;
1600
- const beforeLift = before.holdoutLift ?? null;
1601
- const afterLift = after.holdoutLift ?? null;
1602
- return {
1603
- beforeRunId: before.runId,
1604
- afterRunId: after.runId,
1605
- beforeTimestamp: before.timestamp,
1606
- afterTimestamp: after.timestamp,
1607
- beforeGateDecision: before.gateDecision ?? null,
1608
- afterGateDecision: after.gateDecision ?? null,
1609
- beforeHoldoutLift: beforeLift,
1610
- afterHoldoutLift: afterLift,
1611
- holdoutLiftDelta: beforeLift !== null && afterLift !== null ? afterLift - beforeLift : null,
1612
- beforeTotalCostUsd: before.totalCostUsd,
1613
- afterTotalCostUsd: after.totalCostUsd,
1614
- totalCostUsdDelta: after.totalCostUsd - before.totalCostUsd,
1615
- beforeTotalDurationMs: before.totalDurationMs,
1616
- afterTotalDurationMs: after.totalDurationMs,
1617
- totalDurationMsDelta: after.totalDurationMs - before.totalDurationMs,
1618
- baselineDiff,
1619
- winnersDiff
1620
- };
1621
- }
1422
+ const beforeWinner = winnerOf(before);
1423
+ const afterWinner = winnerOf(after);
1424
+ const baselineDiff = before.baseline && after.baseline ? diffGenerations(before.baseline, after.baseline) : null;
1425
+ const winnersDiff = beforeWinner && afterWinner ? diffGenerations(beforeWinner, afterWinner) : null;
1426
+ const beforeLift = before.holdoutLift ?? null;
1427
+ const afterLift = after.holdoutLift ?? null;
1428
+ return {
1429
+ beforeRunId: before.runId,
1430
+ afterRunId: after.runId,
1431
+ beforeTimestamp: before.timestamp,
1432
+ afterTimestamp: after.timestamp,
1433
+ beforeGateDecision: before.gateDecision ?? null,
1434
+ afterGateDecision: after.gateDecision ?? null,
1435
+ beforeHoldoutLift: beforeLift,
1436
+ afterHoldoutLift: afterLift,
1437
+ holdoutLiftDelta: beforeLift !== null && afterLift !== null ? afterLift - beforeLift : null,
1438
+ beforeTotalCostUsd: before.totalCostUsd,
1439
+ afterTotalCostUsd: after.totalCostUsd,
1440
+ totalCostUsdDelta: after.totalCostUsd - before.totalCostUsd,
1441
+ beforeTotalDurationMs: before.totalDurationMs,
1442
+ afterTotalDurationMs: after.totalDurationMs,
1443
+ totalDurationMsDelta: after.totalDurationMs - before.totalDurationMs,
1444
+ baselineDiff,
1445
+ winnersDiff
1446
+ };
1447
+ }
1448
+ /**
1449
+ * Within-run baseline → winning-generation diff. The natural "what did the
1450
+ * improvement loop produce" view for a single run. Returns null when the
1451
+ * run never reached a generation past baseline (errored early, or the gate
1452
+ * shipped the baseline as-is).
1453
+ */
1622
1454
  function diffRunBaselineToWinner(run) {
1623
- if (!run.baseline) return null;
1624
- const winner = winnerOf(run);
1625
- if (!winner || winner.index === run.baseline.index) return null;
1626
- return diffGenerations(run.baseline, winner);
1455
+ if (!run.baseline) return null;
1456
+ const winner = winnerOf(run);
1457
+ if (!winner || winner.index === run.baseline.index) return null;
1458
+ return diffGenerations(run.baseline, winner);
1627
1459
  }
1628
-
1629
- // src/contract/intake/agent-trace.ts
1460
+ //#endregion
1461
+ //#region src/contract/intake/agent-trace.ts
1630
1462
  function rangeLines(r) {
1631
- return Math.max(0, r.end_line - r.start_line + 1);
1463
+ return Math.max(0, r.end_line - r.start_line + 1);
1632
1464
  }
1465
+ /**
1466
+ * Build a commit → provenance index from Agent Trace records. Multiple records
1467
+ * for the same revision are merged. Records without `vcs.revision` are skipped
1468
+ * (the SHA is the join key — without it there is nothing to correlate against).
1469
+ */
1633
1470
  function parseAgentTrace(records) {
1634
- const acc = /* @__PURE__ */ new Map();
1635
- for (const record of records) {
1636
- const sha = record.vcs?.revision;
1637
- if (!sha) continue;
1638
- let a = acc.get(sha);
1639
- if (!a) {
1640
- a = {
1641
- models: /* @__PURE__ */ new Set(),
1642
- tools: /* @__PURE__ */ new Set(),
1643
- files: /* @__PURE__ */ new Set(),
1644
- conversationCount: 0,
1645
- lineCount: 0,
1646
- humanInvolved: false
1647
- };
1648
- acc.set(sha, a);
1649
- }
1650
- if (record.tool?.name) a.tools.add(record.tool.name);
1651
- for (const file of record.files ?? []) {
1652
- a.files.add(file.path);
1653
- for (const conv of file.conversations ?? []) {
1654
- a.conversationCount += 1;
1655
- for (const range of conv.ranges ?? []) {
1656
- const contributor = range.contributor ?? conv.contributor;
1657
- a.lineCount += rangeLines(range);
1658
- if (!contributor) continue;
1659
- if (contributor.type === "human" || contributor.type === "mixed") {
1660
- a.humanInvolved = true;
1661
- }
1662
- if ((contributor.type === "ai" || contributor.type === "mixed") && contributor.model_id) {
1663
- a.models.add(contributor.model_id);
1664
- }
1665
- }
1666
- }
1667
- }
1668
- }
1669
- const index = /* @__PURE__ */ new Map();
1670
- for (const [sha, a] of acc) {
1671
- index.set(sha, {
1672
- commitSha: sha,
1673
- aiModels: [...a.models].sort(),
1674
- tools: [...a.tools].sort(),
1675
- conversationCount: a.conversationCount,
1676
- fileCount: a.files.size,
1677
- lineCount: a.lineCount,
1678
- humanInvolved: a.humanInvolved
1679
- });
1680
- }
1681
- return index;
1682
- }
1471
+ const acc = /* @__PURE__ */ new Map();
1472
+ for (const record of records) {
1473
+ const sha = record.vcs?.revision;
1474
+ if (!sha) continue;
1475
+ let a = acc.get(sha);
1476
+ if (!a) {
1477
+ a = {
1478
+ models: /* @__PURE__ */ new Set(),
1479
+ tools: /* @__PURE__ */ new Set(),
1480
+ files: /* @__PURE__ */ new Set(),
1481
+ conversationCount: 0,
1482
+ lineCount: 0,
1483
+ humanInvolved: false
1484
+ };
1485
+ acc.set(sha, a);
1486
+ }
1487
+ if (record.tool?.name) a.tools.add(record.tool.name);
1488
+ for (const file of record.files ?? []) {
1489
+ a.files.add(file.path);
1490
+ for (const conv of file.conversations ?? []) {
1491
+ a.conversationCount += 1;
1492
+ for (const range of conv.ranges ?? []) {
1493
+ const contributor = range.contributor ?? conv.contributor;
1494
+ a.lineCount += rangeLines(range);
1495
+ if (!contributor) continue;
1496
+ if (contributor.type === "human" || contributor.type === "mixed") a.humanInvolved = true;
1497
+ if ((contributor.type === "ai" || contributor.type === "mixed") && contributor.model_id) a.models.add(contributor.model_id);
1498
+ }
1499
+ }
1500
+ }
1501
+ }
1502
+ const index = /* @__PURE__ */ new Map();
1503
+ for (const [sha, a] of acc) index.set(sha, {
1504
+ commitSha: sha,
1505
+ aiModels: [...a.models].sort(),
1506
+ tools: [...a.tools].sort(),
1507
+ conversationCount: a.conversationCount,
1508
+ fileCount: a.files.size,
1509
+ lineCount: a.lineCount,
1510
+ humanInvolved: a.humanInvolved
1511
+ });
1512
+ return index;
1513
+ }
1514
+ /**
1515
+ * Partition runs by the AI model(s) that authored the code at each run's
1516
+ * `commitSha`. Feed `byModel.get(modelId)` to `analyzeRuns`, or compare two
1517
+ * model cohorts via `analyzeRuns({ runs: a, baselineRuns: b })` for a lift CI
1518
+ * on "model A's code vs model B's code".
1519
+ */
1683
1520
  function partitionRunsByAuthoringModel(runs, index) {
1684
- const byModel = /* @__PURE__ */ new Map();
1685
- const unattributed = [];
1686
- for (const run of runs) {
1687
- const provenance = index.get(run.commitSha);
1688
- if (!provenance || provenance.aiModels.length === 0) {
1689
- unattributed.push(run);
1690
- continue;
1691
- }
1692
- for (const model of provenance.aiModels) {
1693
- const cohort = byModel.get(model) ?? [];
1694
- cohort.push(run);
1695
- byModel.set(model, cohort);
1696
- }
1697
- }
1698
- return { byModel, unattributed };
1699
- }
1700
-
1701
- // src/contract/intake/feedback-table.ts
1521
+ const byModel = /* @__PURE__ */ new Map();
1522
+ const unattributed = [];
1523
+ for (const run of runs) {
1524
+ const provenance = index.get(run.commitSha);
1525
+ if (!provenance || provenance.aiModels.length === 0) {
1526
+ unattributed.push(run);
1527
+ continue;
1528
+ }
1529
+ for (const model of provenance.aiModels) {
1530
+ const cohort = byModel.get(model) ?? [];
1531
+ cohort.push(run);
1532
+ byModel.set(model, cohort);
1533
+ }
1534
+ }
1535
+ return {
1536
+ byModel,
1537
+ unattributed
1538
+ };
1539
+ }
1540
+ //#endregion
1541
+ //#region src/contract/intake/feedback-table.ts
1702
1542
  function fromFeedbackTable(opts) {
1703
- const { ratings, meta = [], scale, emitRaterScores = true } = opts;
1704
- const metaByRun = new Map(meta.map((m) => [m.runId, m]));
1705
- const normalise = (rating) => {
1706
- if (typeof rating === "boolean") return rating ? 1 : 0;
1707
- if (!Number.isFinite(rating)) return Number.NaN;
1708
- if (!scale) return rating;
1709
- const { min, max } = scale;
1710
- if (max === min) return rating;
1711
- return (rating - min) / (max - min);
1712
- };
1713
- const byRun = /* @__PURE__ */ new Map();
1714
- for (const row of ratings) {
1715
- const list = byRun.get(row.runId) ?? [];
1716
- list.push(row);
1717
- byRun.set(row.runId, list);
1718
- }
1719
- const runs = [];
1720
- const raterScores = [];
1721
- for (const [runId, rowsForRun] of byRun) {
1722
- const normalised = rowsForRun.map((r) => ({ rater: r.rater, score: normalise(r.rating) })).filter((r) => Number.isFinite(r.score));
1723
- if (normalised.length === 0) continue;
1724
- const meanScore = normalised.reduce((s, r) => s + r.score, 0) / normalised.length;
1725
- const runMeta = metaByRun.get(runId) ?? { runId };
1726
- const judgeScores = {
1727
- perJudge: Object.fromEntries(normalised.map((r) => [r.rater, { rating: r.score }])),
1728
- perDimMean: { rating: meanScore },
1729
- composite: meanScore
1730
- };
1731
- const splitTag = runMeta.splitTag ?? "holdout";
1732
- const outcome = {
1733
- ...splitTag === "holdout" ? { holdoutScore: meanScore } : { searchScore: meanScore },
1734
- raw: Object.fromEntries(normalised.map((r) => [`rater:${r.rater}`, r.score])),
1735
- judgeScores
1736
- };
1737
- const costUsd = runMeta.costUsd ?? null;
1738
- runs.push({
1739
- runId,
1740
- experimentId: runMeta.experimentId ?? "feedback-corpus",
1741
- candidateId: runMeta.candidateId ?? runId,
1742
- seed: 0,
1743
- model: runMeta.model ?? "unknown@unknown",
1744
- promptHash: runMeta.promptHash ?? "sha256:unknown",
1745
- configHash: runMeta.configHash ?? "sha256:unknown",
1746
- commitSha: runMeta.commitSha ?? "unknown",
1747
- wallMs: runMeta.wallMs ?? 0,
1748
- costUsd,
1749
- costProvenance: costUsd === null ? { kind: "uncaptured", usd: null } : { kind: "observed", usd: costUsd },
1750
- tokenUsage: { input: 0, output: 0 },
1751
- terminalOutcome: "unknown",
1752
- outcome,
1753
- splitTag,
1754
- scenarioId: runMeta.scenarioId ?? runId
1755
- });
1756
- if (emitRaterScores) {
1757
- for (const r of normalised) raterScores.push({ runId, rater: r.rater, score: r.score });
1758
- }
1759
- }
1760
- return { runs, raterScores };
1761
- }
1762
-
1763
- // src/contract/intake/otel-spans.ts
1764
- var TASK_SCORE_ATTR_KEYS = [
1765
- "gen_ai.evaluation.score.value",
1766
- "tangle.task.score",
1767
- "eval.score",
1768
- "tangle.score"
1543
+ const { ratings, meta = [], scale, emitRaterScores = true } = opts;
1544
+ const metaByRun = new Map(meta.map((m) => [m.runId, m]));
1545
+ const normalise = (rating) => {
1546
+ if (typeof rating === "boolean") return rating ? 1 : 0;
1547
+ if (!Number.isFinite(rating)) return NaN;
1548
+ if (!scale) return rating;
1549
+ const { min, max } = scale;
1550
+ if (max === min) return rating;
1551
+ return (rating - min) / (max - min);
1552
+ };
1553
+ const byRun = /* @__PURE__ */ new Map();
1554
+ for (const row of ratings) {
1555
+ const list = byRun.get(row.runId) ?? [];
1556
+ list.push(row);
1557
+ byRun.set(row.runId, list);
1558
+ }
1559
+ const runs = [];
1560
+ const raterScores = [];
1561
+ for (const [runId, rowsForRun] of byRun) {
1562
+ const normalised = rowsForRun.map((r) => ({
1563
+ rater: r.rater,
1564
+ score: normalise(r.rating)
1565
+ })).filter((r) => Number.isFinite(r.score));
1566
+ if (normalised.length === 0) continue;
1567
+ const meanScore = normalised.reduce((s, r) => s + r.score, 0) / normalised.length;
1568
+ const runMeta = metaByRun.get(runId) ?? { runId };
1569
+ const judgeScores = {
1570
+ perJudge: Object.fromEntries(normalised.map((r) => [r.rater, { rating: r.score }])),
1571
+ perDimMean: { rating: meanScore },
1572
+ composite: meanScore
1573
+ };
1574
+ const splitTag = runMeta.splitTag ?? "holdout";
1575
+ const outcome = {
1576
+ ...splitTag === "holdout" ? { holdoutScore: meanScore } : { searchScore: meanScore },
1577
+ raw: Object.fromEntries(normalised.map((r) => [`rater:${r.rater}`, r.score])),
1578
+ judgeScores
1579
+ };
1580
+ const costUsd = runMeta.costUsd ?? null;
1581
+ runs.push({
1582
+ runId,
1583
+ experimentId: runMeta.experimentId ?? "feedback-corpus",
1584
+ candidateId: runMeta.candidateId ?? runId,
1585
+ seed: 0,
1586
+ model: runMeta.model ?? "unknown@unknown",
1587
+ promptHash: runMeta.promptHash ?? "sha256:unknown",
1588
+ configHash: runMeta.configHash ?? "sha256:unknown",
1589
+ commitSha: runMeta.commitSha ?? "unknown",
1590
+ wallMs: runMeta.wallMs ?? 0,
1591
+ costUsd,
1592
+ costProvenance: costUsd === null ? {
1593
+ kind: "uncaptured",
1594
+ usd: null
1595
+ } : {
1596
+ kind: "observed",
1597
+ usd: costUsd
1598
+ },
1599
+ tokenUsage: {
1600
+ input: 0,
1601
+ output: 0
1602
+ },
1603
+ terminalOutcome: "unknown",
1604
+ outcome,
1605
+ splitTag,
1606
+ scenarioId: runMeta.scenarioId ?? runId
1607
+ });
1608
+ if (emitRaterScores) for (const r of normalised) raterScores.push({
1609
+ runId,
1610
+ rater: r.rater,
1611
+ score: r.score
1612
+ });
1613
+ }
1614
+ return {
1615
+ runs,
1616
+ raterScores
1617
+ };
1618
+ }
1619
+ //#endregion
1620
+ //#region src/contract/intake/otel-spans.ts
1621
+ /**
1622
+ * # `intake/otel-spans` — OTel `TraceSpanEvent[]` → `RunRecord[]`.
1623
+ *
1624
+ * Turns an existing observability stream into the substrate-canonical
1625
+ * `RunRecord` shape so consumers with logs but no eval discipline can
1626
+ * call `analyzeRuns()` against their production traffic immediately.
1627
+ *
1628
+ * Pivot rule: spans are grouped by `tangle.runId` (the same attribute the
1629
+ * hosted-tier wire format uses) or, when absent, by `traceId`. One group
1630
+ * becomes one `RunRecord`. The root span (no `parentSpanId`) supplies:
1631
+ *
1632
+ * - `runId` (the group key)
1633
+ * - `wallMs` from `endTimeUnixNano - startTimeUnixNano`
1634
+ * - `model` from `gen_ai.request.model` / `llm.model` / `tangle.model`
1635
+ * - task failure class and detail from explicit `tangle.task.*` attributes
1636
+ * - cost from `cost.usd` / `gen_ai.usage.cost_usd` / `tangle.cost.usd`
1637
+ * - token usage from model-call input, output, cache-read, and cache-write
1638
+ * attributes without double-counting aggregate parent spans
1639
+ * - task quality from an explicit `scoreForRun` callback or a designated
1640
+ * evaluation attribute on a root / `EVALUATOR` span; `outcome.raw`
1641
+ * collects every numeric attribute without promoting it to task quality.
1642
+ *
1643
+ * Errored tool, model, and child-agent spans contribute to execution-error
1644
+ * counts. Root process, guardrail, evaluator, propagated parent, and unknown
1645
+ * errors retain separate counters. Only one failed root can set
1646
+ * `RunRecord.terminalOutcome` and `RunRecord.terminalFailureReason`; a child
1647
+ * error cannot become a task failure.
1648
+ */
1649
+ const TASK_SCORE_ATTR_KEYS = [
1650
+ "gen_ai.evaluation.score.value",
1651
+ "tangle.task.score",
1652
+ "eval.score",
1653
+ "tangle.score"
1769
1654
  ];
1770
- var MODEL_KEYS = ["tangle.model", ...LLM_MODEL_ATTR_KEYS, "model"];
1771
- var PROMPT_HASH_KEYS = ["tangle.prompt_hash", "prompt.hash"];
1772
- var CONFIG_HASH_KEYS = ["tangle.config_hash", "config.hash"];
1655
+ const MODEL_KEYS = [
1656
+ "tangle.model",
1657
+ ...LLM_MODEL_ATTR_KEYS,
1658
+ "model"
1659
+ ];
1660
+ const PROMPT_HASH_KEYS = ["tangle.prompt_hash", "prompt.hash"];
1661
+ const CONFIG_HASH_KEYS = ["tangle.config_hash", "config.hash"];
1773
1662
  function fromOtelSpans(opts) {
1774
- const { spans, defaultSplit = "holdout", experimentId = "otel-corpus" } = opts;
1775
- const grouped = groupSpans(spans);
1776
- const runs = [];
1777
- for (const [groupKey, groupSpans2] of grouped) {
1778
- const root = findRoot(groupSpans2);
1779
- if (!root) continue;
1780
- const measurements = summarizeExecutionMeasurements(
1781
- groupSpans2.map((span) => ({
1782
- id: span.spanId,
1783
- ...span.parentSpanId ? { parentId: span.parentSpanId } : {},
1784
- attributes: span.attributes,
1785
- modelCall: isExplicitModelCall(span),
1786
- aggregate: isExplicitAggregate(span)
1787
- }))
1788
- );
1789
- const callSpanIds = new Set(measurements.callSpanIds);
1790
- const callSpans = groupSpans2.filter((span) => callSpanIds.has(span.spanId));
1791
- const wallMs = unixNanoDurationMs(root.startTimeUnixNano, root.endTimeUnixNano);
1792
- const model = readAttrString(callSpans, MODEL_KEYS) ?? readAttrString(groupSpans2, MODEL_KEYS) ?? "unknown@unknown";
1793
- const capturedCost = (measurements.cost.complete ? measurements.cost.value : void 0) ?? measurements.aggregate?.costUsd;
1794
- const costUsd = capturedCost ?? null;
1795
- const scenarioId = readConsistentScenarioId(groupKey, groupSpans2) ?? groupKey;
1796
- const promptHash = readAttrString(groupSpans2, PROMPT_HASH_KEYS) ?? "sha256:unknown";
1797
- const configHash = readAttrString(groupSpans2, CONFIG_HASH_KEYS) ?? "sha256:unknown";
1798
- const score = resolveTaskScore(groupKey, groupSpans2, opts.scoreForRun);
1799
- const taskFailure = readTaskFailureLabels(
1800
- groupSpans2.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span)),
1801
- `fromOtelSpans: run '${groupKey}'`
1802
- );
1803
- const rawNumeric = collectNumericAttrs(groupSpans2);
1804
- const errorSummary = summarizeTraceErrors(
1805
- groupSpans2.map((span) => ({
1806
- id: spanIdentity(span),
1807
- ...span.parentSpanId ? { parentId: parentIdentity(span) } : {},
1808
- role: errorRoleForSpan(span),
1809
- error: span.status?.code === "ERROR",
1810
- processRoot: !span.parentSpanId && isTerminalRootCandidate(span)
1811
- }))
1812
- );
1813
- rawNumeric.error_span_count = errorSummary.total;
1814
- rawNumeric.execution_error_count = errorSummary.execution;
1815
- rawNumeric.process_error_count = errorSummary.process;
1816
- rawNumeric.guardrail_error_count = errorSummary.guardrail;
1817
- rawNumeric.judge_error_count = errorSummary.evaluation;
1818
- rawNumeric.propagated_error_count = errorSummary.propagated;
1819
- rawNumeric.unclassified_error_count = errorSummary.unclassified;
1820
- rawNumeric.llm_span_count = measurements.modelCallCount;
1821
- if (measurements.cost.value !== void 0 && !measurements.cost.complete) {
1822
- rawNumeric.partial_observed_cost_usd = measurements.cost.value;
1823
- }
1824
- recordAggregateMeasurements(rawNumeric, measurements.aggregate);
1825
- const judgeScores = score !== void 0 ? {
1826
- perJudge: { "otel-derived": { score } },
1827
- perDimMean: { score },
1828
- composite: score
1829
- } : void 0;
1830
- const terminalOutcome = terminalOutcomeFromRoots(groupSpans2);
1831
- const failedRoot = terminalOutcome === "failed" ? groupSpans2.find(
1832
- (span) => !span.parentSpanId && isTerminalRootCandidate(span) && span.status?.code === "ERROR"
1833
- ) : void 0;
1834
- const outcome = {
1835
- raw: rawNumeric,
1836
- ...judgeScores ? { judgeScores } : {}
1837
- };
1838
- if (score !== void 0) {
1839
- if (defaultSplit === "holdout") outcome.holdoutScore = score;
1840
- else outcome.searchScore = score;
1841
- }
1842
- runs.push({
1843
- runId: groupKey,
1844
- experimentId,
1845
- candidateId: root.attributes["tangle.candidateId"] ?? "otel-default",
1846
- seed: 0,
1847
- model,
1848
- promptHash,
1849
- configHash,
1850
- commitSha: root.attributes["tangle.commit_sha"] ?? "unknown",
1851
- wallMs,
1852
- costUsd,
1853
- costProvenance: capturedCost === void 0 ? { kind: "uncaptured", usd: null } : { kind: "observed", usd: capturedCost },
1854
- tokenUsage: measurements.tokenUsage,
1855
- terminalOutcome,
1856
- ...failedRoot ? { terminalFailureReason: failedRoot.status?.message ?? failedRoot.name } : {},
1857
- outcome,
1858
- ...taskFailure,
1859
- splitTag: defaultSplit,
1860
- scenarioId
1861
- });
1862
- }
1863
- return runs;
1663
+ const { spans, defaultSplit = "holdout", experimentId = "otel-corpus" } = opts;
1664
+ const grouped = groupSpans(spans);
1665
+ const runs = [];
1666
+ for (const [groupKey, groupSpans] of grouped) {
1667
+ const root = findRoot(groupSpans);
1668
+ if (!root) continue;
1669
+ const measurements = summarizeExecutionMeasurements(groupSpans.map((span) => ({
1670
+ id: span.spanId,
1671
+ ...span.parentSpanId ? { parentId: span.parentSpanId } : {},
1672
+ attributes: span.attributes,
1673
+ modelCall: isExplicitModelCall(span),
1674
+ aggregate: isExplicitAggregate(span)
1675
+ })));
1676
+ const callSpanIds = new Set(measurements.callSpanIds);
1677
+ const callSpans = groupSpans.filter((span) => callSpanIds.has(span.spanId));
1678
+ const wallMs = unixNanoDurationMs(root.startTimeUnixNano, root.endTimeUnixNano);
1679
+ const model = readAttrString(callSpans, MODEL_KEYS) ?? readAttrString(groupSpans, MODEL_KEYS) ?? "unknown@unknown";
1680
+ const capturedCost = (measurements.cost.complete ? measurements.cost.value : void 0) ?? measurements.aggregate?.costUsd;
1681
+ const costUsd = capturedCost ?? null;
1682
+ const scenarioId = readConsistentScenarioId(groupKey, groupSpans) ?? groupKey;
1683
+ const promptHash = readAttrString(groupSpans, PROMPT_HASH_KEYS) ?? "sha256:unknown";
1684
+ const configHash = readAttrString(groupSpans, CONFIG_HASH_KEYS) ?? "sha256:unknown";
1685
+ const score = resolveTaskScore(groupKey, groupSpans, opts.scoreForRun);
1686
+ const taskFailure = readTaskFailureLabels(groupSpans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span)), `fromOtelSpans: run '${groupKey}'`);
1687
+ const rawNumeric = collectNumericAttrs(groupSpans);
1688
+ const errorSummary = summarizeTraceErrors(groupSpans.map((span) => ({
1689
+ id: spanIdentity(span),
1690
+ ...span.parentSpanId ? { parentId: parentIdentity(span) } : {},
1691
+ role: errorRoleForSpan(span),
1692
+ error: span.status?.code === "ERROR",
1693
+ processRoot: !span.parentSpanId && isTerminalRootCandidate(span)
1694
+ })));
1695
+ rawNumeric.error_span_count = errorSummary.total;
1696
+ rawNumeric.execution_error_count = errorSummary.execution;
1697
+ rawNumeric.process_error_count = errorSummary.process;
1698
+ rawNumeric.guardrail_error_count = errorSummary.guardrail;
1699
+ rawNumeric.judge_error_count = errorSummary.evaluation;
1700
+ rawNumeric.propagated_error_count = errorSummary.propagated;
1701
+ rawNumeric.unclassified_error_count = errorSummary.unclassified;
1702
+ rawNumeric.llm_span_count = measurements.modelCallCount;
1703
+ if (measurements.cost.value !== void 0 && !measurements.cost.complete) rawNumeric.partial_observed_cost_usd = measurements.cost.value;
1704
+ recordAggregateMeasurements(rawNumeric, measurements.aggregate);
1705
+ const judgeScores = score !== void 0 ? {
1706
+ perJudge: { "otel-derived": { score } },
1707
+ perDimMean: { score },
1708
+ composite: score
1709
+ } : void 0;
1710
+ const terminalOutcome = terminalOutcomeFromRoots(groupSpans);
1711
+ const failedRoot = terminalOutcome === "failed" ? groupSpans.find((span) => !span.parentSpanId && isTerminalRootCandidate(span) && span.status?.code === "ERROR") : void 0;
1712
+ const outcome = {
1713
+ raw: rawNumeric,
1714
+ ...judgeScores ? { judgeScores } : {}
1715
+ };
1716
+ if (score !== void 0) if (defaultSplit === "holdout") outcome.holdoutScore = score;
1717
+ else outcome.searchScore = score;
1718
+ runs.push({
1719
+ runId: groupKey,
1720
+ experimentId,
1721
+ candidateId: root.attributes["tangle.candidateId"] ?? "otel-default",
1722
+ seed: 0,
1723
+ model,
1724
+ promptHash,
1725
+ configHash,
1726
+ commitSha: root.attributes["tangle.commit_sha"] ?? "unknown",
1727
+ wallMs,
1728
+ costUsd,
1729
+ costProvenance: capturedCost === void 0 ? {
1730
+ kind: "uncaptured",
1731
+ usd: null
1732
+ } : {
1733
+ kind: "observed",
1734
+ usd: capturedCost
1735
+ },
1736
+ tokenUsage: measurements.tokenUsage,
1737
+ terminalOutcome,
1738
+ ...failedRoot ? { terminalFailureReason: failedRoot.status?.message ?? failedRoot.name } : {},
1739
+ outcome,
1740
+ ...taskFailure,
1741
+ splitTag: defaultSplit,
1742
+ scenarioId
1743
+ });
1744
+ }
1745
+ return runs;
1864
1746
  }
1865
1747
  function terminalOutcomeFromRoots(spans) {
1866
- const roots = spans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span));
1867
- if (roots.length !== 1) return "unknown";
1868
- if (roots[0].status?.code === "OK") return "succeeded";
1869
- if (roots[0].status?.code === "ERROR") return "failed";
1870
- return "unknown";
1748
+ const roots = spans.filter((span) => !span.parentSpanId && isTerminalRootCandidate(span));
1749
+ if (roots.length !== 1) return "unknown";
1750
+ if (roots[0].status?.code === "OK") return "succeeded";
1751
+ if (roots[0].status?.code === "ERROR") return "failed";
1752
+ return "unknown";
1871
1753
  }
1872
1754
  function isTerminalRootCandidate(span) {
1873
- const role = errorRoleForSpan(span);
1874
- if (role === "LLM" || role === "TOOL" || role === "GUARDRAIL" || role === "EVALUATOR") {
1875
- return false;
1876
- }
1877
- return true;
1755
+ const role = errorRoleForSpan(span);
1756
+ if (role === "LLM" || role === "TOOL" || role === "GUARDRAIL" || role === "EVALUATOR") return false;
1757
+ return true;
1878
1758
  }
1879
1759
  function readSpanKind(span) {
1880
- return readAttrString([span], [...SPAN_KIND_ATTR_KEYS, "span.kind"])?.toUpperCase();
1760
+ return readAttrString([span], [...SPAN_KIND_ATTR_KEYS, "span.kind"])?.toUpperCase();
1881
1761
  }
1882
1762
  function errorRoleForSpan(span) {
1883
- return classifyOtlpSpanRole({
1884
- kind: readSpanKind(span),
1885
- name: span.name,
1886
- attributes: span.attributes
1887
- });
1763
+ return classifyOtlpSpanRole({
1764
+ kind: readSpanKind(span),
1765
+ name: span.name,
1766
+ attributes: span.attributes
1767
+ });
1888
1768
  }
1889
1769
  function spanIdentity(span) {
1890
- return `${span.traceId}:${span.spanId}`;
1770
+ return `${span.traceId}:${span.spanId}`;
1891
1771
  }
1892
1772
  function parentIdentity(span) {
1893
- return `${span.traceId}:${span.parentSpanId}`;
1773
+ return `${span.traceId}:${span.parentSpanId}`;
1894
1774
  }
1895
1775
  function isExplicitModelCall(span) {
1896
- return isOtlpModelCall({
1897
- kind: readSpanKind(span),
1898
- name: span.name,
1899
- attributes: span.attributes
1900
- });
1776
+ return isOtlpModelCall({
1777
+ kind: readSpanKind(span),
1778
+ name: span.name,
1779
+ attributes: span.attributes
1780
+ });
1901
1781
  }
1902
1782
  function isExplicitAggregate(span) {
1903
- const kind = readSpanKind(span);
1904
- return kind !== void 0 && kind !== "LLM";
1783
+ const kind = readSpanKind(span);
1784
+ return kind !== void 0 && kind !== "LLM";
1905
1785
  }
1906
1786
  function groupSpans(spans) {
1907
- const m = /* @__PURE__ */ new Map();
1908
- for (const span of spans) {
1909
- const key = span["tangle.runId"] ?? span.traceId;
1910
- const list = m.get(key) ?? [];
1911
- list.push(span);
1912
- m.set(key, list);
1913
- }
1914
- return m;
1787
+ const m = /* @__PURE__ */ new Map();
1788
+ for (const span of spans) {
1789
+ const key = span["tangle.runId"] ?? span.traceId;
1790
+ const list = m.get(key) ?? [];
1791
+ list.push(span);
1792
+ m.set(key, list);
1793
+ }
1794
+ return m;
1915
1795
  }
1916
1796
  function findRoot(group) {
1917
- const structuralRoots = group.filter((span) => !span.parentSpanId);
1918
- const terminalRoots = structuralRoots.filter(isTerminalRootCandidate);
1919
- const pool = terminalRoots.length > 0 ? terminalRoots : structuralRoots.length > 0 ? structuralRoots : group;
1920
- return orderSpans(pool)[0];
1797
+ const structuralRoots = group.filter((span) => !span.parentSpanId);
1798
+ const terminalRoots = structuralRoots.filter(isTerminalRootCandidate);
1799
+ return orderSpans(terminalRoots.length > 0 ? terminalRoots : structuralRoots.length > 0 ? structuralRoots : group)[0];
1921
1800
  }
1922
1801
  function readAttrString(spans, keys) {
1923
- for (const span of spans) {
1924
- for (const key of keys) {
1925
- const v = span.attributes[key];
1926
- if (typeof v === "string" && v.length > 0) return v;
1927
- }
1928
- }
1929
- return void 0;
1802
+ for (const span of spans) for (const key of keys) {
1803
+ const v = span.attributes[key];
1804
+ if (typeof v === "string" && v.length > 0) return v;
1805
+ }
1930
1806
  }
1931
1807
  function readConsistentScenarioId(runId, spans) {
1932
- const values = /* @__PURE__ */ new Set();
1933
- for (const span of spans) {
1934
- const topLevel = span["tangle.scenarioId"];
1935
- if (typeof topLevel === "string" && topLevel.length > 0) values.add(topLevel);
1936
- const attribute = span.attributes["tangle.scenarioId"];
1937
- if (typeof attribute === "string" && attribute.length > 0) values.add(attribute);
1938
- }
1939
- if (values.size > 1) {
1940
- throw new ValidationError(
1941
- `fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`
1942
- );
1943
- }
1944
- return values.values().next().value;
1808
+ const values = /* @__PURE__ */ new Set();
1809
+ for (const span of spans) {
1810
+ const topLevel = span["tangle.scenarioId"];
1811
+ if (typeof topLevel === "string" && topLevel.length > 0) values.add(topLevel);
1812
+ const attribute = span.attributes["tangle.scenarioId"];
1813
+ if (typeof attribute === "string" && attribute.length > 0) values.add(attribute);
1814
+ }
1815
+ if (values.size > 1) throw new ValidationError(`fromOtelSpans: conflicting scenario ids for run '${runId}': ${[...values].sort().join(", ")}`);
1816
+ return values.values().next().value;
1945
1817
  }
1946
1818
  function resolveTaskScore(runId, spans, scoreForRun) {
1947
- const orderedSpans = orderSpans(spans);
1948
- const sources = [];
1949
- if (scoreForRun) {
1950
- const supplied = scoreForRun(runId, orderedSpans);
1951
- if (supplied !== void 0) {
1952
- if (typeof supplied !== "number" || !Number.isFinite(supplied)) {
1953
- throw new ValidationError(
1954
- `fromOtelSpans: scoreForRun returned a non-finite number for run '${runId}'`
1955
- );
1956
- }
1957
- sources.push({ label: "scoreForRun", value: supplied });
1958
- }
1959
- }
1960
- for (const span of orderedSpans) {
1961
- const role = errorRoleForSpan(span);
1962
- if (span.parentSpanId && role !== "EVALUATOR") continue;
1963
- if (role === "EVALUATOR" && span.status?.code === "ERROR") continue;
1964
- for (const key of TASK_SCORE_ATTR_KEYS) {
1965
- if (!Object.hasOwn(span.attributes, key)) continue;
1966
- sources.push({
1967
- label: `span '${span.spanId}' attribute '${key}'`,
1968
- value: parseTaskScoreAttribute(runId, span.spanId, key, span.attributes[key])
1969
- });
1970
- }
1971
- }
1972
- if (sources.length === 0) return void 0;
1973
- sources.sort((left, right) => left.label.localeCompare(right.label));
1974
- const score = sources[0].value;
1975
- if (sources.some((source) => source.value !== score)) {
1976
- const details = sources.map((source) => `${source.label}=${source.value}`).join(", ");
1977
- throw new ValidationError(
1978
- `fromOtelSpans: conflicting task-quality scores for run '${runId}': ${details}`
1979
- );
1980
- }
1981
- return score;
1819
+ const orderedSpans = orderSpans(spans);
1820
+ const sources = [];
1821
+ if (scoreForRun) {
1822
+ const supplied = scoreForRun(runId, orderedSpans);
1823
+ if (supplied !== void 0) {
1824
+ if (typeof supplied !== "number" || !Number.isFinite(supplied)) throw new ValidationError(`fromOtelSpans: scoreForRun returned a non-finite number for run '${runId}'`);
1825
+ sources.push({
1826
+ label: "scoreForRun",
1827
+ value: supplied
1828
+ });
1829
+ }
1830
+ }
1831
+ for (const span of orderedSpans) {
1832
+ const role = errorRoleForSpan(span);
1833
+ if (span.parentSpanId && role !== "EVALUATOR") continue;
1834
+ if (role === "EVALUATOR" && span.status?.code === "ERROR") continue;
1835
+ for (const key of TASK_SCORE_ATTR_KEYS) {
1836
+ if (!Object.hasOwn(span.attributes, key)) continue;
1837
+ sources.push({
1838
+ label: `span '${span.spanId}' attribute '${key}'`,
1839
+ value: parseTaskScoreAttribute(runId, span.spanId, key, span.attributes[key])
1840
+ });
1841
+ }
1842
+ }
1843
+ if (sources.length === 0) return void 0;
1844
+ sources.sort((left, right) => left.label.localeCompare(right.label));
1845
+ const score = sources[0].value;
1846
+ if (sources.some((source) => source.value !== score)) throw new ValidationError(`fromOtelSpans: conflicting task-quality scores for run '${runId}': ${sources.map((source) => `${source.label}=${source.value}`).join(", ")}`);
1847
+ return score;
1982
1848
  }
1983
1849
  function parseTaskScoreAttribute(runId, spanId, key, value) {
1984
- const source = `span '${spanId}' attribute '${key}'`;
1985
- if (typeof value === "string") {
1986
- if (value.trim().length === 0) {
1987
- throw new ValidationError(
1988
- `fromOtelSpans: ${source} is blank for run '${runId}'; task quality must be finite`
1989
- );
1990
- }
1991
- const parsed = Number(value);
1992
- if (Number.isFinite(parsed)) return parsed;
1993
- } else if (typeof value === "number" && Number.isFinite(value)) {
1994
- return value;
1995
- }
1996
- throw new ValidationError(
1997
- `fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`
1998
- );
1850
+ const source = `span '${spanId}' attribute '${key}'`;
1851
+ if (typeof value === "string") {
1852
+ if (value.trim().length === 0) throw new ValidationError(`fromOtelSpans: ${source} is blank for run '${runId}'; task quality must be finite`);
1853
+ const parsed = Number(value);
1854
+ if (Number.isFinite(parsed)) return parsed;
1855
+ } else if (typeof value === "number" && Number.isFinite(value)) return value;
1856
+ throw new ValidationError(`fromOtelSpans: ${source} is not a finite task-quality score for run '${runId}'`);
1999
1857
  }
2000
1858
  function orderSpans(spans) {
2001
- return [...spans].sort(
2002
- (left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId)
2003
- );
1859
+ return [...spans].sort((left, right) => compareUnixNano(left.startTimeUnixNano, right.startTimeUnixNano) || left.spanId.localeCompare(right.spanId));
2004
1860
  }
2005
1861
  function parseUnixNano(value) {
2006
- try {
2007
- return BigInt(value);
2008
- } catch {
2009
- throw new ValidationError(`fromOtelSpans: invalid Unix nanosecond timestamp '${value}'`);
2010
- }
1862
+ try {
1863
+ return BigInt(value);
1864
+ } catch {
1865
+ throw new ValidationError(`fromOtelSpans: invalid Unix nanosecond timestamp '${value}'`);
1866
+ }
2011
1867
  }
2012
1868
  function compareUnixNano(left, right) {
2013
- const leftValue = parseUnixNano(left);
2014
- const rightValue = parseUnixNano(right);
2015
- return leftValue < rightValue ? -1 : leftValue > rightValue ? 1 : 0;
1869
+ const leftValue = parseUnixNano(left);
1870
+ const rightValue = parseUnixNano(right);
1871
+ return leftValue < rightValue ? -1 : leftValue > rightValue ? 1 : 0;
2016
1872
  }
2017
1873
  function unixNanoDurationMs(start, end) {
2018
- const delta = parseUnixNano(end) - parseUnixNano(start);
2019
- if (delta <= 0n) return 0;
2020
- const wholeMs = delta / 1000000n;
2021
- const fractionalMs = delta % 1000000n;
2022
- const value = Number(wholeMs) + Number(fractionalMs) / 1e6;
2023
- if (!Number.isSafeInteger(Number(wholeMs))) {
2024
- throw new ValidationError("fromOtelSpans: span duration exceeds the safe millisecond range");
2025
- }
2026
- return value;
1874
+ const delta = parseUnixNano(end) - parseUnixNano(start);
1875
+ if (delta <= 0n) return 0;
1876
+ const wholeMs = delta / 1000000n;
1877
+ const fractionalMs = delta % 1000000n;
1878
+ const value = Number(wholeMs) + Number(fractionalMs) / 1e6;
1879
+ if (!Number.isSafeInteger(Number(wholeMs))) throw new ValidationError("fromOtelSpans: span duration exceeds the safe millisecond range");
1880
+ return value;
2027
1881
  }
2028
1882
  function collectNumericAttrs(spans) {
2029
- const raw = {};
2030
- for (const span of spans) {
2031
- for (const [k, v] of Object.entries(span.attributes)) {
2032
- if (typeof v === "number" && Number.isFinite(v)) raw[k] = v;
2033
- }
2034
- }
2035
- return raw;
2036
- }
2037
- export {
2038
- FileSystemOutcomeStore,
2039
- InMemoryOutcomeStore,
2040
- REFERENCE_EQUIVALENCE_INPUT_LIMITS,
2041
- REFERENCE_EQUIVALENCE_JUDGE_VERSION,
2042
- SelfImproveRunError,
2043
- analyzeRuns,
2044
- buildDefaultAnalystRegistry,
2045
- buildEvidenceVector,
2046
- campaignSplitDigest,
2047
- compareOptimizationMethods,
2048
- composeGate,
2049
- createChatClient,
2050
- createReferenceEquivalenceJudge,
2051
- defaultProductionGate,
2052
- defineAgentEval,
2053
- diffGenerations,
2054
- diffRunBaselineToWinner,
2055
- diffRuns,
2056
- evalReportingSuite,
2057
- evaluatePairedMeasurements,
2058
- externalTextOptimizationMethod,
2059
- fromClaudeCodeSession,
2060
- fromCodexSession,
2061
- fromFeedbackTable,
2062
- fromKimiCodeSession,
2063
- fromOpenCodeSession,
2064
- fromOtelSpans,
2065
- fromPiSession,
2066
- fromPigraphSession,
2067
- fromRunRecordDir,
2068
- fsCampaignStorage,
2069
- gepaOptimizationMethod,
2070
- heldOutGate,
2071
- inMemoryCampaignStorage,
2072
- llmJudge,
2073
- measuredComparisonFromCandidateExperiment,
2074
- observeCodeAgentSession,
2075
- paretoPolicy,
2076
- paretoSignificanceGate,
2077
- parseAgentTrace,
2078
- parseCodeAgentJsonl,
2079
- partitionRunsByAuthoringModel,
2080
- runCampaign,
2081
- runCandidateExperiment,
2082
- runEval,
2083
- runImprovementLoop,
2084
- runReferenceEquivalenceJudge,
2085
- sealCandidateBenchmarkSuite,
2086
- sealCandidateBenchmarkTask,
2087
- sealCandidateExperiment,
2088
- selfImprove,
2089
- skillOptOptimizationMethod,
2090
- summarizeExecution,
2091
- verifyCandidateBenchmarkSuite,
2092
- verifyCandidateBenchmarkSuiteInputs,
2093
- verifyCandidateBenchmarkTask,
2094
- verifyCandidateExperiment,
2095
- verifyCandidateExperimentComparison
2096
- };
1883
+ const raw = {};
1884
+ for (const span of spans) for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "number" && Number.isFinite(v)) raw[k] = v;
1885
+ return raw;
1886
+ }
1887
+ //#endregion
1888
+ export { FileSystemOutcomeStore, InMemoryOutcomeStore, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, SelfImproveRunError, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
1889
+
2097
1890
  //# sourceMappingURL=index.js.map