@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
@@ -1,546 +1,83 @@
1
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
2
- interface BudgetSpec {
3
- tokens?: number;
4
- wallMs?: number;
5
- calls?: number;
6
- usd?: number;
7
- }
8
- interface RunOutcome {
9
- score?: number;
10
- pass?: boolean;
11
- failureClass?: FailureClass;
12
- notes?: string;
13
- }
14
- /**
15
- * Layer — optional classification in a nested build workflow.
16
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
17
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
18
- * `app-runtime`: a run of the generated agent against a domain scenario.
19
- * `meta`: any meta-eval (judge replay, correlation analysis).
20
- */
21
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
22
- interface Run {
23
- runId: string;
24
- /**
25
- * Stable identifier of the scenario being executed.
26
- *
27
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
28
- * input WITHOUT this field, substituting a sensible default
29
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
30
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
31
- * keeps the persisted shape unambiguous for downstream filters + aggregations
32
- * while removing the boilerplate of inventing placeholder ids at the call site.
33
- */
34
- scenarioId: string;
35
- variantId?: string;
36
- datasetVersion?: string;
37
- /** Git SHA of agent code at run time. */
38
- codeSha?: string;
39
- /** Hash of the prompt template + any system prompt. */
40
- promptSha?: string;
41
- /** Model id + date + system-prompt hash, concatenated. */
42
- modelFingerprint?: string;
43
- seed?: number;
44
- /** Arbitrary environment markers (shell, docker version, tz). */
45
- envFingerprint?: Record<string, string>;
46
- /** Version of the redaction rules applied to this run. */
47
- redactionVersion?: string;
48
- /** Parent run in a nested build workflow. A builder run's children are
49
- * app-build runs; those children are app-runtime runs. */
50
- parentRunId?: string;
51
- /** Stable project identifier — groups runs across chats + sessions. */
52
- projectId?: string;
53
- /** Chat/conversation identifier within a project. */
54
- chatId?: string;
55
- /** Layer classification — hint for aggregation; not enforced. */
56
- layer?: RunLayer;
57
- startedAt: number;
58
- endedAt?: number;
59
- status: RunStatus;
60
- outcome?: RunOutcome;
61
- budget?: BudgetSpec;
62
- /** Free-form labels for downstream grouping. */
63
- tags?: Record<string, string>;
64
- }
65
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
66
- type SpanStatus = 'ok' | 'error';
67
- interface SpanBase {
68
- spanId: string;
69
- parentSpanId?: string;
70
- runId: string;
71
- kind: SpanKind;
72
- name: string;
73
- startedAt: number;
74
- endedAt?: number;
75
- status?: SpanStatus;
76
- error?: string;
77
- /** Anything not covered by typed fields. Kept deliberately free-form. */
78
- attributes?: Record<string, unknown>;
79
- }
80
- interface Message {
81
- role: 'system' | 'user' | 'assistant' | 'tool';
82
- content: string;
83
- tokens?: number;
84
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
85
- images?: Array<{
86
- artifactId?: string;
87
- url?: string;
88
- mime?: string;
89
- }>;
90
- }
91
- interface LlmSpan extends SpanBase {
92
- kind: 'llm';
93
- model: string;
94
- messages: Message[];
95
- output?: string;
96
- inputTokens?: number;
97
- /** All generated tokens, including the reasoning subset when present. */
98
- outputTokens?: number;
99
- cachedTokens?: number;
100
- cacheWriteTokens?: number;
101
- /** Reasoning-token subset of `outputTokens`. */
102
- reasoningTokens?: number;
103
- costUsd?: number;
104
- finishReason?: string;
105
- }
106
- interface ToolSpan extends SpanBase {
107
- kind: 'tool';
108
- toolName: string;
109
- args: unknown;
110
- /** False when the source observed the call but did not capture its arguments. */
111
- argsCaptured?: boolean;
112
- result?: unknown;
113
- latencyMs?: number;
114
- }
115
- interface RetrievalSpan extends SpanBase {
116
- kind: 'retrieval';
117
- query: string;
118
- hits: Array<{
119
- docId: string;
120
- score: number;
121
- content?: string;
122
- }>;
123
- }
124
- interface JudgeSpan extends SpanBase {
125
- kind: 'judge';
126
- judgeId: string;
127
- /** Span this judgment applies to. */
128
- targetSpanId: string;
129
- dimension: string;
130
- /** Numeric score (free-range; interpretation up to the judge). */
131
- score: number;
132
- rationale?: string;
133
- evidence?: string;
134
- }
135
- interface SandboxSpan extends SpanBase {
136
- kind: 'sandbox';
137
- image?: string;
138
- command?: string;
139
- exitCode?: number;
140
- testsTotal?: number;
141
- testsPassed?: number;
142
- stdoutHash?: string;
143
- stderrHash?: string;
144
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
145
- wallMs?: number;
146
- }
147
- interface GenericSpan extends SpanBase {
148
- kind: 'agent' | 'custom';
149
- }
150
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
151
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
152
- interface TraceEvent {
153
- eventId: string;
154
- runId: string;
155
- spanId?: string;
156
- kind: EventKind;
157
- timestamp: number;
158
- payload: Record<string, unknown>;
159
- }
160
- interface BudgetLedgerEntry {
161
- runId: string;
162
- dimension: keyof BudgetSpec;
163
- limit: number;
164
- consumed: number;
165
- remaining: number;
166
- timestamp: number;
167
- breached: boolean;
168
- /** Span that triggered this entry, if any. */
169
- spanId?: string;
170
- }
171
- interface Artifact {
172
- artifactId: string;
173
- runId: string;
174
- spanId?: string;
175
- contentType: string;
176
- sizeBytes: number;
177
- /** sha256 in hex. */
178
- hash: string;
179
- /** External storage URL (R2, S3, filesystem path). */
180
- storageUrl?: string;
181
- /** Inline content for small blobs — keep under ~64KB. */
182
- inlineContent?: string;
183
- }
184
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
185
-
186
- interface RunFilter {
187
- scenarioId?: string;
188
- variantId?: string;
189
- status?: RunStatus;
190
- since?: number;
191
- until?: number;
192
- tag?: {
193
- key: string;
194
- value: string;
195
- };
196
- parentRunId?: string;
197
- projectId?: string;
198
- chatId?: string;
199
- layer?: RunLayer;
200
- }
201
- interface SpanFilter {
202
- runId?: string;
203
- parentSpanId?: string;
204
- kind?: SpanKind;
205
- name?: string;
206
- toolName?: string;
207
- judgeId?: string;
208
- since?: number;
209
- until?: number;
210
- }
211
- interface EventFilter {
212
- runId?: string;
213
- spanId?: string;
214
- kind?: EventKind;
215
- since?: number;
216
- until?: number;
217
- }
218
- interface TraceStore {
219
- appendRun(run: Run): Promise<void>;
220
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
221
- appendSpan(span: Span): Promise<void>;
222
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
223
- appendEvent(event: TraceEvent): Promise<void>;
224
- appendArtifact(artifact: Artifact): Promise<void>;
225
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
226
- getRun(runId: string): Promise<Run | undefined>;
227
- listRuns(filter?: RunFilter): Promise<Run[]>;
228
- spans(filter?: SpanFilter): Promise<Span[]>;
229
- events(filter?: EventFilter): Promise<TraceEvent[]>;
230
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
231
- artifacts(runId: string): Promise<Artifact[]>;
232
- }
233
-
234
- /**
235
- * TraceEmitter — hierarchical span builder that auto-parents using an
236
- * internal stack. One emitter per Run; emitters do NOT share state.
237
- *
238
- * Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)
239
- * return a `SpanHandle` with `.end()` / `.fail()` so callers don't
240
- * have to thread spanIds manually. For async workflows that can't use
241
- * the stack (e.g. fan-out parallel calls), pass `parentSpanId`
242
- * explicitly.
243
- */
244
-
245
- interface SpanHandle<S extends Span = Span> {
246
- span: S;
247
- end(patch?: Partial<S>): Promise<void>;
248
- fail(error: string | Error, patch?: Partial<S>): Promise<void>;
249
- }
250
- interface RunCompleteHookContext {
251
- runId: string;
252
- emitter: TraceEmitter;
253
- store: TraceStore;
254
- /** Outcome the caller passed to `endRun` (undefined for `abortRun`). */
255
- outcome?: RunOutcome;
256
- /** Final run status. */
257
- status: 'completed' | 'failed' | 'aborted';
258
- }
259
- type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void;
260
- interface TraceEmitterOptions {
261
- runId?: string;
262
- /** Inject a clock for deterministic tests. */
263
- now?: () => number;
264
- /** Inject an id generator for deterministic tests. */
265
- id?: () => string;
266
- /**
267
- * Hooks fired after `endRun` / `abortRun` writes the final run state.
268
- * Designed for trace-analyst auto-execution, integrity assertions, and
269
- * outbound notifications. Hooks run sequentially in the order supplied.
270
- *
271
- * By default a hook that throws is swallowed and logged as a `note` event
272
- * on the run — auto-orchestration must not crash the underlying flow.
273
- * Set `hookErrors: 'throw'` to propagate.
274
- */
275
- onRunComplete?: RunCompleteHook[];
276
- /** `'swallow'` (default) | `'throw'`. */
277
- hookErrors?: 'swallow' | 'throw';
278
- }
279
- declare class TraceEmitter {
280
- private store;
281
- private stack;
282
- private _runId;
283
- private now;
284
- private id;
285
- private hooks;
286
- private hookErrors;
287
- constructor(store: TraceStore, options?: TraceEmitterOptions);
288
- get runId(): string;
289
- get traceStore(): TraceStore;
290
- /** Append a hook after construction (e.g. attach the trace analyst). */
291
- addRunCompleteHook(hook: RunCompleteHook): void;
292
- /**
293
- * Begin a Run.
294
- *
295
- * `scenarioId` is required on the persisted Run shape — every Run downstream
296
- * gets a non-empty scenarioId so filters and aggregations stay simple — but
297
- * the INPUT here accepts it as optional. When omitted, startRun substitutes
298
- * a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so
299
- * runtime / operator / meta-eval runs that have no curated-scenario corpus
300
- * to anchor to don't have to invent placeholder strings at the call site.
301
- */
302
- startRun(run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & {
303
- scenarioId?: string;
304
- }): Promise<Run>;
305
- endRun(outcome?: RunOutcome): Promise<void>;
306
- abortRun(reason: string): Promise<void>;
307
- private runHooks;
308
- span<S extends Span = Span>(init: {
309
- kind: SpanKind;
310
- name: string;
311
- parentSpanId?: string;
312
- attributes?: Record<string, unknown>;
313
- } & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>): Promise<SpanHandle<S>>;
314
- private handle;
315
- private pop;
316
- llm(init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<LlmSpan>>;
317
- tool(init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<ToolSpan>>;
318
- retrieval(init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<RetrievalSpan>>;
319
- recordJudge(verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>): Promise<JudgeSpan>;
320
- sandbox(init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<SandboxSpan>>;
321
- emit(event: {
322
- kind: EventKind;
323
- spanId?: string;
324
- payload?: Record<string, unknown>;
325
- }): Promise<TraceEvent>;
326
- recordBudget(entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & {
327
- timestamp?: number;
328
- }): Promise<BudgetLedgerEntry>;
329
- recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact>;
330
- /**
331
- * Runs `fn` inside a span; auto-ends on success, auto-fails on throw.
332
- * Returns the fn's return value. Use this for the 95% case.
333
- */
334
- within<T>(init: Parameters<TraceEmitter['span']>[0], fn: (handle: SpanHandle) => Promise<T>): Promise<T>;
335
- }
336
-
337
- /**
338
- * SandboxHarness — executes a scenario in an isolated environment and
339
- * emits a rich SandboxSpan into the trace.
340
- *
341
- * Two built-in drivers:
342
- * - `SubprocessSandboxDriver` — spawn in a local cwd with env vars.
343
- * Fast, no dependencies, fine for unit tests and most CI gates.
344
- * - `DockerSandboxDriver` — lifted from tangle-router's sandbox path;
345
- * shells out to `docker run`. Stronger isolation, slower startup.
346
- *
347
- * Consumers implement `SandboxDriver` for custom backends (Firecracker,
348
- * Cloudflare sandbox product, etc.). The harness doesn't care which.
349
- */
350
-
351
- interface HarnessConfig {
352
- /** Setup command (e.g. "pnpm install"). Non-zero exit fails the run. */
353
- setupCommand?: string;
354
- /** Run command (e.g. "pnpm build"). */
355
- runCommand?: string;
356
- /** Test command (e.g. "pnpm test --run"). Drives the test count + pass count. */
357
- testCommand?: string;
358
- /** Absolute cwd for the subprocess driver. Ignored by docker driver. */
359
- cwd?: string;
360
- /** Max wall-clock per phase in ms. Default 10 minutes. */
361
- timeoutMs?: number;
362
- /**
363
- * Cap on captured stdout+stderr bytes per phase. A runaway process can
364
- * otherwise grow the in-memory buffer without bound. Once hit, further
365
- * output is dropped and `outputTruncated` is set. Default 16 MiB.
366
- */
367
- maxOutputBytes?: number;
368
- /** Image for the docker driver. */
369
- image?: string;
370
- /** Extra env vars (validated; shell-escaped). */
371
- env?: Record<string, string>;
372
- /** Parser for the test output — maps stdout/stderr/exit code → pass count. */
373
- testParser?: TestOutputParser;
374
- }
375
- interface TestOutputParser {
376
- id: string;
377
- parse(stdout: string, stderr: string, exitCode: number): {
378
- testsTotal: number;
379
- testsPassed: number;
380
- } | undefined;
381
- }
382
- interface SandboxResult {
383
- phase: 'setup' | 'run' | 'test';
384
- exitCode: number;
385
- stdout: string;
386
- stderr: string;
387
- wallMs: number;
388
- testsTotal?: number;
389
- testsPassed?: number;
390
- /**
391
- * True when the process was killed because it exceeded `timeoutMs`. A
392
- * SIGKILLed child can still close with exit code 0; callers MUST treat
393
- * a timed-out phase as a hard failure regardless of `exitCode`, never
394
- * as a pass. `undefined`/`false` means the process completed on its own.
395
- */
396
- killedByTimeout?: boolean;
397
- /**
398
- * True when captured stdout/stderr hit `maxOutputBytes` and further
399
- * output was dropped. The result is still returned (the process was
400
- * not killed for this), but downstream parsers see truncated text.
401
- */
402
- outputTruncated?: boolean;
403
- }
404
- interface SandboxDriver {
405
- id: string;
406
- exec(phase: SandboxResult['phase'], command: string, config: HarnessConfig): Promise<SandboxResult>;
407
- }
408
- interface SandboxHarnessResult {
409
- passed: boolean;
410
- setup?: SandboxResult;
411
- run?: SandboxResult;
412
- test?: SandboxResult;
413
- totalWallMs: number;
414
- /** Final score — 0 when no tests; otherwise testsPassed/testsTotal. */
415
- score: number;
416
- }
417
-
418
- /**
419
- * TestGradedScenario — a scenario whose score comes from a test suite.
420
- *
421
- * This is the SWE-bench pattern generalized. The scenario ships:
422
- * - fixture data (setup instructions)
423
- * - a test command the harness runs
424
- * - optional assertion overrides
425
- *
426
- * The runner emits a run, delegates to SandboxHarness, records the
427
- * outcome, and returns a structured verdict. Consumers bind their own
428
- * agent execution to this contract.
429
- */
430
-
431
- interface TestGradedScenario {
432
- id: string;
433
- description?: string;
434
- harness: HarnessConfig;
435
- /** Optional pass threshold in 0..1 (default 1.0 = all tests must pass). */
436
- passThreshold?: number;
437
- /** Provenance for dataset tracking. */
438
- datasetVersion?: string;
439
- /** Free-form tags (difficulty, category, etc.). */
440
- tags?: Record<string, string>;
441
- }
442
- interface TestGradedRunResult {
443
- runId: string;
444
- scenario: TestGradedScenario;
445
- harness: SandboxHarnessResult;
446
- pass: boolean;
447
- score: number;
448
- failureClass?: FailureClass;
449
- }
450
-
451
- /**
452
- * BuilderSession — ties a builder-of-builders workflow together.
453
- *
454
- * Models agent-builder's shape: Project → Chat → Edit → Ship → App →
455
- * AppAgent. Each layer is a Run (linked via parentRunId). The
456
- * framework-enforced invariants:
457
- *
458
- * - One Project → many Chats; chatId scopes runs within a project.
459
- * - One Chat = one builder Run with `layer='builder'`.
460
- * - One Ship = one child Run with `layer='app-build'` + SandboxHarness.
461
- * - One AppScenario = one grandchild Run with `layer='app-runtime'`.
462
- *
463
- * Consumers obtain a BuilderSession, call `startChat`, drive the
464
- * builder agent (emitting spans), and call `ship` / `runAppScenario`
465
- * as the workflow progresses. The session reconstructs itself from
466
- * trace data via `resume(store, projectId)`.
467
- */
468
-
1
+ import { f as Run } from "../schema-BtVldJ3T.js";
2
+ import { s as TraceStore } from "../store-CT9YIIve.js";
3
+ import { i as TraceEmitter } from "../emitter-DGQGoLyj.js";
4
+ import { l as SandboxHarnessResult, n as TestGradedRunResult, o as HarnessConfig, r as TestGradedScenario, s as SandboxDriver } from "../test-graded-scenario-D1TaI2va.js";
5
+ //#region src/builder-eval/builder-session.d.ts
469
6
  interface BuilderSessionInit {
470
- projectId: string;
471
- chatId?: string;
472
- /** Free-form: user's task description, project name, etc. Stored on the builder Run. */
473
- tags?: Record<string, string>;
7
+ projectId: string;
8
+ chatId?: string;
9
+ /** Free-form: user's task description, project name, etc. Stored on the builder Run. */
10
+ tags?: Record<string, string>;
474
11
  }
475
12
  interface ShipOptions {
476
- harness: HarnessConfig;
477
- driver?: SandboxDriver;
478
- /** scenarioId of this app-build run. Defaults to `${projectId}/build`. */
479
- scenarioId?: string;
13
+ harness: HarnessConfig;
14
+ driver?: SandboxDriver;
15
+ /** scenarioId of this app-build run. Defaults to `${projectId}/build`. */
16
+ scenarioId?: string;
480
17
  }
481
18
  interface RunAppScenarioOptions {
482
- scenario: TestGradedScenario;
483
- /** Harness driver override; defaults to the one the session was created with. */
484
- driver?: SandboxDriver;
19
+ scenario: TestGradedScenario;
20
+ /** Harness driver override; defaults to the one the session was created with. */
21
+ driver?: SandboxDriver;
485
22
  }
486
23
  declare class BuilderSession {
487
- private store;
488
- private builderEmitter;
489
- readonly projectId: string;
490
- readonly chatId: string;
491
- private builderRunId?;
492
- private lastBuildRunId?;
493
- private defaultDriver?;
494
- constructor(store: TraceStore, init: BuilderSessionInit, driver?: SandboxDriver);
495
- /** Start the builder (L0) run for this chat. Returns the runId. */
496
- startChat(scenarioId?: string): Promise<string>;
497
- /** The emitter for builder-level spans (edits, LLM calls, tool invocations). */
498
- get emitter(): TraceEmitter;
499
- /**
500
- * Ship the project's generated app: run the sandbox harness as a child
501
- * Run (`layer='app-build'`). Returns the build result + runId.
502
- */
503
- ship(options: ShipOptions): Promise<{
504
- runId: string;
505
- result: SandboxHarnessResult;
506
- }>;
507
- /**
508
- * Run a domain scenario against the just-built app as a grandchild Run
509
- * (`layer='app-runtime'`). The `ship` call must precede this so the
510
- * parent is set correctly; if no build exists yet the session attaches
511
- * directly to the builder run (useful for prototypes).
512
- */
513
- runAppScenario(options: RunAppScenarioOptions): Promise<TestGradedRunResult>;
514
- /** Record an end-of-chat meta score (judge verdict on whether the builder
515
- * served the user's intent). Accepts a numeric score + optional rationale. */
516
- recordMetaScore(score: number, rationale?: string): Promise<void>;
517
- /** Close the builder Run with a final outcome. */
518
- endChat(outcome: {
519
- pass: boolean;
520
- score?: number;
521
- notes?: string;
522
- }): Promise<void>;
523
- /**
524
- * Inline app-runtime run — for cases where the "scenario" isn't a
525
- * SWE-bench-style test suite but a live agent interaction (LLM chat,
526
- * domain flow). Returns an emitter bound to a fresh Run in the
527
- * `app-runtime` layer; caller emits spans inside and calls
528
- * `.endRun()` with the final verdict.
529
- */
530
- startAppRuntime(scenarioId: string): Promise<TraceEmitter>;
531
- /**
532
- * Lightweight "ship marker" — record an app-build Run with a caller-
533
- * provided verdict. Use when there isn't a sandbox harness to run but
534
- * you still want to mark the build state at publish time.
535
- */
536
- recordShipMarker(args: {
537
- pass: boolean;
538
- score: number;
539
- scenarioId?: string;
540
- notes?: string;
541
- }): Promise<string>;
542
- get lastBuildRunIdValue(): string | undefined;
543
- get builderRunIdValue(): string | undefined;
24
+ private store;
25
+ private builderEmitter;
26
+ readonly projectId: string;
27
+ readonly chatId: string;
28
+ private builderRunId?;
29
+ private lastBuildRunId?;
30
+ private defaultDriver?;
31
+ constructor(store: TraceStore, init: BuilderSessionInit, driver?: SandboxDriver);
32
+ /** Start the builder (L0) run for this chat. Returns the runId. */
33
+ startChat(scenarioId?: string): Promise<string>;
34
+ /** The emitter for builder-level spans (edits, LLM calls, tool invocations). */
35
+ get emitter(): TraceEmitter;
36
+ /**
37
+ * Ship the project's generated app: run the sandbox harness as a child
38
+ * Run (`layer='app-build'`). Returns the build result + runId.
39
+ */
40
+ ship(options: ShipOptions): Promise<{
41
+ runId: string;
42
+ result: SandboxHarnessResult;
43
+ }>;
44
+ /**
45
+ * Run a domain scenario against the just-built app as a grandchild Run
46
+ * (`layer='app-runtime'`). The `ship` call must precede this so the
47
+ * parent is set correctly; if no build exists yet the session attaches
48
+ * directly to the builder run (useful for prototypes).
49
+ */
50
+ runAppScenario(options: RunAppScenarioOptions): Promise<TestGradedRunResult>;
51
+ /** Record an end-of-chat meta score (judge verdict on whether the builder
52
+ * served the user's intent). Accepts a numeric score + optional rationale. */
53
+ recordMetaScore(score: number, rationale?: string): Promise<void>;
54
+ /** Close the builder Run with a final outcome. */
55
+ endChat(outcome: {
56
+ pass: boolean;
57
+ score?: number;
58
+ notes?: string;
59
+ }): Promise<void>;
60
+ /**
61
+ * Inline app-runtime run — for cases where the "scenario" isn't a
62
+ * SWE-bench-style test suite but a live agent interaction (LLM chat,
63
+ * domain flow). Returns an emitter bound to a fresh Run in the
64
+ * `app-runtime` layer; caller emits spans inside and calls
65
+ * `.endRun()` with the final verdict.
66
+ */
67
+ startAppRuntime(scenarioId: string): Promise<TraceEmitter>;
68
+ /**
69
+ * Lightweight "ship marker" — record an app-build Run with a caller-
70
+ * provided verdict. Use when there isn't a sandbox harness to run but
71
+ * you still want to mark the build state at publish time.
72
+ */
73
+ recordShipMarker(args: {
74
+ pass: boolean;
75
+ score: number;
76
+ scenarioId?: string;
77
+ notes?: string;
78
+ }): Promise<string>;
79
+ get lastBuildRunIdValue(): string | undefined;
80
+ get builderRunIdValue(): string | undefined;
544
81
  }
545
82
  /**
546
83
  * Reconstruct the most recent BuilderSession state for a given project —
@@ -548,148 +85,99 @@ declare class BuilderSession {
548
85
  * this is how a resumed session finds its place in the edit history.
549
86
  */
550
87
  declare function resumeBuilderSession(store: TraceStore, projectId: string): Promise<{
551
- projectId: string;
552
- chatRuns: Run[];
553
- lastBuilderRun?: Run;
554
- lastBuildRun?: Run;
555
- lastAppRuntimeRuns: Run[];
88
+ projectId: string;
89
+ chatRuns: Run[];
90
+ lastBuilderRun?: Run;
91
+ lastBuildRun?: Run;
92
+ lastAppRuntimeRuns: Run[];
556
93
  }>;
557
-
558
- /**
559
- * Three-layer evaluation — the canonical scoring breakdown for
560
- * builder-of-builders workflows.
561
- *
562
- * meta_score: did the builder understand + satisfy user intent?
563
- * (judge verdict attached to the builder run)
564
- * build_score: did the generated scaffold build + pass its own tests?
565
- * (outcome.score on the app-build child run)
566
- * runtime_score: did the generated agent pass its domain scenarios?
567
- * (mean outcome.score over app-runtime grandchild runs)
568
- *
569
- * Returns a structured report per project. The cross-layer correlation
570
- * is the highest-leverage signal the framework computes — if
571
- * meta_score doesn't predict runtime_score, the builder's self-scoring
572
- * is broken.
573
- *
574
- * Scaffold-only mode: when a project has no `app-runtime` runs (e.g. a
575
- * scaffold-builder eval that grades compose + build without driving a
576
- * runtime scenario), `kind` is `'scaffold-only'` and `complete` measures
577
- * meta + build only. Consumers can tell the two apart without having to
578
- * interpret null-runtime as either "not yet computed" or "N/A for this
579
- * project shape".
580
- */
581
-
94
+ //#endregion
95
+ //#region src/builder-eval/three-layer-eval.d.ts
582
96
  type ProjectKind = 'full' | 'scaffold-only';
583
97
  interface ThreeLayerProjectReport {
584
- projectId: string;
585
- /**
586
- * `'full'` when the project has at least one `app-runtime` run;
587
- * `'scaffold-only'` when it only has meta + build layers. Lets
588
- * downstream consumers treat a null runtime score as expected
589
- * (scaffold-only) vs. missing (full, pipeline broke).
590
- */
591
- kind: ProjectKind;
592
- builderRunId?: string;
593
- /** Judge-verdict score on the builder run (0..1 after normalization). */
594
- metaScore: number | null;
595
- buildRunId?: string;
596
- /** 0..1 from the sandbox harness (testsPassed / testsTotal). */
597
- buildScore: number | null;
598
- appRuntimeRunIds: string[];
599
- /** Mean of outcome.score over app-runtime runs, 0..1. Always null in scaffold-only mode. */
600
- runtimeScore: number | null;
601
- runtimePassRate: number | null;
602
- /**
603
- * Layer-aware completeness:
604
- * - `kind='full'`: all three layers scored
605
- * - `kind='scaffold-only'`: meta + build scored (runtime not applicable)
606
- */
607
- complete: boolean;
98
+ projectId: string;
99
+ /**
100
+ * `'full'` when the project has at least one `app-runtime` run;
101
+ * `'scaffold-only'` when it only has meta + build layers. Lets
102
+ * downstream consumers treat a null runtime score as expected
103
+ * (scaffold-only) vs. missing (full, pipeline broke).
104
+ */
105
+ kind: ProjectKind;
106
+ builderRunId?: string;
107
+ /** Judge-verdict score on the builder run (0..1 after normalization). */
108
+ metaScore: number | null;
109
+ buildRunId?: string;
110
+ /** 0..1 from the sandbox harness (testsPassed / testsTotal). */
111
+ buildScore: number | null;
112
+ appRuntimeRunIds: string[];
113
+ /** Mean of outcome.score over app-runtime runs, 0..1. Always null in scaffold-only mode. */
114
+ runtimeScore: number | null;
115
+ runtimePassRate: number | null;
116
+ /**
117
+ * Layer-aware completeness:
118
+ * - `kind='full'`: all three layers scored
119
+ * - `kind='scaffold-only'`: meta + build scored (runtime not applicable)
120
+ */
121
+ complete: boolean;
608
122
  }
609
123
  declare function scoreProject(store: TraceStore, projectId: string): Promise<ThreeLayerProjectReport>;
610
124
  /** Aggregate scoring across every project in a corpus. */
611
125
  declare function scoreAllProjects(store: TraceStore): Promise<ThreeLayerProjectReport[]>;
612
-
613
- /**
614
- * Meta-eval correlation — the highest-leverage signal in the framework.
615
- *
616
- * Given a corpus of three-layer project reports, compute how well each
617
- * pair of layers correlates. The question we care about most:
618
- *
619
- * Does `metaScore` (what the builder thinks it did) predict
620
- * `runtimeScore` (what the user actually gets)?
621
- *
622
- * If r < ~0.4, the builder's self-scoring is broken — it's optimizing
623
- * for something other than real-world success. If r > 0.7, meta_score
624
- * is a usable proxy and can drive CI gates cheaply.
625
- *
626
- * Non-parametric rank correlation (Spearman) is also reported because
627
- * meta scores are often ordinal-ish.
628
- */
629
-
126
+ //#endregion
127
+ //#region src/builder-eval/correlation.d.ts
630
128
  interface LayerCorrelation {
631
- n: number;
632
- pearson: number;
633
- spearman: number;
129
+ n: number;
130
+ pearson: number;
131
+ spearman: number;
634
132
  }
635
133
  interface CorrelationReport {
636
- /** Pairs present in the corpus (layers with ≥ 2 matched data points). */
637
- metaVsBuild?: LayerCorrelation;
638
- metaVsRuntime?: LayerCorrelation;
639
- buildVsRuntime?: LayerCorrelation;
640
- /** Number of complete projects (all 3 scores present). */
641
- completeProjects: number;
134
+ /** Pairs present in the corpus (layers with ≥ 2 matched data points). */
135
+ metaVsBuild?: LayerCorrelation;
136
+ metaVsRuntime?: LayerCorrelation;
137
+ buildVsRuntime?: LayerCorrelation;
138
+ /** Number of complete projects (all 3 scores present). */
139
+ completeProjects: number;
642
140
  }
643
141
  declare function correlateLayers(reports: ThreeLayerProjectReport[]): CorrelationReport;
644
-
645
- /**
646
- * ProjectRegistry — project-level aggregation over the trace corpus.
647
- *
648
- * Thin reader over TraceStore that answers the questions a chat-first,
649
- * resumable UI needs:
650
- * - listProjects() → project IDs with latest activity
651
- * - projectTimeline(id) → chats + builds + runtime runs, chronological
652
- * - projectChats(id) → chat-level summaries (turn count, outcome)
653
- *
654
- * All queries are pure reads; no state duplication.
655
- */
656
-
142
+ //#endregion
143
+ //#region src/builder-eval/project-registry.d.ts
657
144
  interface ProjectSummary {
658
- projectId: string;
659
- chatCount: number;
660
- buildCount: number;
661
- appRuntimeCount: number;
662
- lastActivityAt: number;
663
- latestChatId?: string;
664
- latestOutcome?: {
665
- pass: boolean;
666
- score?: number;
667
- };
145
+ projectId: string;
146
+ chatCount: number;
147
+ buildCount: number;
148
+ appRuntimeCount: number;
149
+ lastActivityAt: number;
150
+ latestChatId?: string;
151
+ latestOutcome?: {
152
+ pass: boolean;
153
+ score?: number;
154
+ };
668
155
  }
669
156
  interface ChatSummary {
670
- chatId: string;
671
- projectId: string;
672
- builderRunId: string;
673
- startedAt: number;
674
- endedAt?: number;
675
- status: Run['status'];
676
- outcome?: Run['outcome'];
677
- /** Counts of spans emitted during the chat. */
678
- llmTurns?: number;
679
- toolCalls?: number;
680
- buildRunId?: string;
681
- appRuntimeRunIds: string[];
157
+ chatId: string;
158
+ projectId: string;
159
+ builderRunId: string;
160
+ startedAt: number;
161
+ endedAt?: number;
162
+ status: Run['status'];
163
+ outcome?: Run['outcome'];
164
+ /** Counts of spans emitted during the chat. */
165
+ llmTurns?: number;
166
+ toolCalls?: number;
167
+ buildRunId?: string;
168
+ appRuntimeRunIds: string[];
682
169
  }
683
170
  interface ProjectTimelineEntry {
684
- run: Run;
685
- layerBucket: 'chat' | 'build' | 'runtime' | 'other';
171
+ run: Run;
172
+ layerBucket: 'chat' | 'build' | 'runtime' | 'other';
686
173
  }
687
174
  declare class ProjectRegistry {
688
- private store;
689
- constructor(store: TraceStore);
690
- listProjects(): Promise<ProjectSummary[]>;
691
- projectTimeline(projectId: string): Promise<ProjectTimelineEntry[]>;
692
- projectChats(projectId: string): Promise<ChatSummary[]>;
693
- }
694
-
695
- export { BuilderSession, type BuilderSessionInit, type ChatSummary, type CorrelationReport, type LayerCorrelation, type ProjectKind, ProjectRegistry, type ProjectSummary, type ProjectTimelineEntry, type RunAppScenarioOptions, type ShipOptions, type ThreeLayerProjectReport, correlateLayers, resumeBuilderSession, scoreAllProjects, scoreProject };
175
+ private store;
176
+ constructor(store: TraceStore);
177
+ listProjects(): Promise<ProjectSummary[]>;
178
+ projectTimeline(projectId: string): Promise<ProjectTimelineEntry[]>;
179
+ projectChats(projectId: string): Promise<ChatSummary[]>;
180
+ }
181
+ //#endregion
182
+ export { BuilderSession, BuilderSessionInit, ChatSummary, CorrelationReport, LayerCorrelation, ProjectKind, ProjectRegistry, ProjectSummary, ProjectTimelineEntry, RunAppScenarioOptions, ShipOptions, ThreeLayerProjectReport, correlateLayers, resumeBuilderSession, scoreAllProjects, scoreProject };
183
+ //# sourceMappingURL=index.d.ts.map