@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,1741 @@
1
+ import { o as ReplayError } from "./errors-8YnH8WlF.js";
2
+ import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
3
+ import { s as validateRunRecord } from "./run-record-BuoE80Dq.js";
4
+ import { a as providerFromBaseUrl, i as defaultProviderRedactor } from "./raw-provider-sink-BQd7mzyT.js";
5
+ import { LLM_INPUT_TOKENS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, OPENINFERENCE_SPAN_KIND, TOOL_NAME, applyLlmSpanOtlpAttributes } from "./trace-attributes.js";
6
+ import { S as traceSpanKindToOpenInferenceKind, _ as spanEpochMillis, b as classifyOtlpSpanRole, d as compareSpanTime, h as projectOtlpFlatLine, p as firstStringAttr, x as isOtlpModelCall, y as applyToolSpanOtlpAttributes } from "./tools-BmuN627J.js";
7
+ import { t as analyzeTraces } from "./analyst-LsnNpSkm.js";
8
+ import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "./task-failure-attributes-CQZlB3et.js";
9
+ import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
10
+ import { readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
11
+ import { join } from "node:path";
12
+ //#region src/trace-analyst/hook.ts
13
+ const DEFAULT_QUESTION = "Summarise what happened in this run. Surface any failure modes, surprising findings, or evidence that the run's verdict is wrong.";
14
+ function traceAnalystOnRunComplete(opts) {
15
+ return async (ctx) => {
16
+ if (opts.shouldRun && !opts.shouldRun(ctx)) return;
17
+ const source = opts.analyze.source;
18
+ if (source === void 0) {
19
+ await ctx.store.appendEvent({
20
+ eventId: `analyst-skip-${ctx.runId}`,
21
+ runId: ctx.runId,
22
+ kind: "log",
23
+ timestamp: Date.now(),
24
+ payload: {
25
+ source: "trace_analyst_hook",
26
+ reason: "no source configured"
27
+ }
28
+ });
29
+ return;
30
+ }
31
+ const result = await analyzeTraces({ question: opts.question ?? DEFAULT_QUESTION }, {
32
+ ...opts.analyze,
33
+ source
34
+ });
35
+ if (opts.save) await opts.save(result, ctx);
36
+ if (opts.gateOn && !opts.gateOn(result, ctx)) await ctx.store.appendEvent({
37
+ eventId: `analyst-gate-${ctx.runId}`,
38
+ runId: ctx.runId,
39
+ kind: "log",
40
+ timestamp: Date.now(),
41
+ payload: {
42
+ source: "trace_analyst_hook",
43
+ reason: "analyst_gate_failed",
44
+ findings: result.findings
45
+ }
46
+ });
47
+ };
48
+ }
49
+ //#endregion
50
+ //#region src/trace-analyst/insights.ts
51
+ const DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
52
+ "and",
53
+ "advanced",
54
+ "app",
55
+ "build",
56
+ "create",
57
+ "easy",
58
+ "expert",
59
+ "extreme",
60
+ "for",
61
+ "from",
62
+ "hard",
63
+ "implementation",
64
+ "integrate",
65
+ "medium",
66
+ "project",
67
+ "task",
68
+ "the",
69
+ "this",
70
+ "with",
71
+ "workflow"
72
+ ]);
73
+ function tokenizeDomainWords(value) {
74
+ return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
75
+ }
76
+ function inferDomainKeywords(suite) {
77
+ const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
78
+ const source = [
79
+ suite.name,
80
+ suite.collectionId ?? "",
81
+ ...suite.tasks.flatMap((task) => [
82
+ task.id,
83
+ task.name,
84
+ task.prompt ?? "",
85
+ task.difficulty ?? "",
86
+ ...task.tags ?? [],
87
+ ...task.gaps ?? []
88
+ ])
89
+ ].join(" ");
90
+ const counts = /* @__PURE__ */ new Map();
91
+ for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
92
+ return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
93
+ }
94
+ function domainEvidencePattern(keywords) {
95
+ const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
96
+ return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
97
+ }
98
+ function describeTraceInsightScope(suite) {
99
+ const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
100
+ const tags = /* @__PURE__ */ new Map();
101
+ for (const task of suite.tasks) for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
102
+ const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
103
+ if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
104
+ return `${taskLabel} across ${[...new Set(suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value)))].join(", ") || "the selected benchmark scope"}.`;
105
+ }
106
+ function planTraceInsightQuestions(input) {
107
+ const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
108
+ const hasMultipleShots = input.suite.tasks.some((task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap)));
109
+ const questions = [
110
+ {
111
+ id: "execution-path",
112
+ question: "What did the worker actually do before the first meaningful implementation edit?",
113
+ why: "Separates grounded execution from polished but shallow output."
114
+ },
115
+ {
116
+ id: "research-grounding",
117
+ question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
118
+ why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
119
+ },
120
+ {
121
+ id: "domain-proof",
122
+ question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
123
+ why: "Keeps product-quality claims tied to concrete evidence."
124
+ },
125
+ {
126
+ id: "root-cause",
127
+ question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
128
+ why: "Turns trace observations into actionable ownership."
129
+ },
130
+ {
131
+ id: "evidence-quality",
132
+ question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
133
+ why: "Prevents unsupported customer-report conclusions."
134
+ }
135
+ ];
136
+ if (hasMultipleShots) questions.push({
137
+ id: "reviewer-lift",
138
+ question: "Where did reviewer feedback improve score, stall, or regress across shots?",
139
+ why: "Shows whether the driver loop is learning or merely repeating work."
140
+ });
141
+ if (hasFailures) questions.push({
142
+ id: "optimization-targets",
143
+ question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
144
+ why: "Connects benchmark evidence to the optimization loop."
145
+ });
146
+ return questions;
147
+ }
148
+ function buildTraceInsightContext(input) {
149
+ return {
150
+ suite: input.suite,
151
+ scope: describeTraceInsightScope(input.suite),
152
+ keywords: inferDomainKeywords(input.suite),
153
+ questions: planTraceInsightQuestions(input),
154
+ panel: defaultTraceInsightPanel(),
155
+ findings: input.findings ?? [],
156
+ agent: input.agent ?? null,
157
+ totals: input.totals ?? null
158
+ };
159
+ }
160
+ function scoreTraceInsightReadiness(context) {
161
+ const failedTasks = context.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied");
162
+ const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
163
+ const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
164
+ const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
165
+ const gates = [
166
+ {
167
+ id: "domain-context",
168
+ label: "Domain context inferred",
169
+ passed: context.keywords.length > 0,
170
+ severity: "high",
171
+ detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
172
+ },
173
+ {
174
+ id: "panel-coverage",
175
+ label: "Analyst panel planned",
176
+ passed: context.panel.length >= 4 && context.questions.length >= 5,
177
+ severity: "high",
178
+ detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
179
+ },
180
+ {
181
+ id: "failure-coverage",
182
+ label: "Failures mapped to findings",
183
+ passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= .5,
184
+ severity: "critical",
185
+ detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
186
+ },
187
+ {
188
+ id: "gap-evidence",
189
+ label: "Task gaps captured",
190
+ passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= .5,
191
+ severity: "medium",
192
+ detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
193
+ }
194
+ ];
195
+ const penalty = gates.reduce((sum, gate) => {
196
+ if (gate.passed) return sum;
197
+ if (gate.severity === "critical") return sum + 35;
198
+ if (gate.severity === "high") return sum + 20;
199
+ if (gate.severity === "medium") return sum + 10;
200
+ return sum + 5;
201
+ }, 0);
202
+ const score = Math.max(0, Math.min(1, 1 - penalty / 100));
203
+ return {
204
+ score,
205
+ grade: score >= .9 ? "external-ready" : score >= .7 ? "internal-review" : "raw-analysis",
206
+ gates
207
+ };
208
+ }
209
+ function defaultTraceInsightPanel() {
210
+ return [
211
+ {
212
+ id: "trace-forensics",
213
+ name: "Trace Forensics",
214
+ responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
215
+ },
216
+ {
217
+ id: "root-cause",
218
+ name: "Root Cause",
219
+ responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
220
+ },
221
+ {
222
+ id: "optimization",
223
+ name: "Optimization",
224
+ responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
225
+ },
226
+ {
227
+ id: "external-evidence",
228
+ name: "External Evidence",
229
+ responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
230
+ }
231
+ ];
232
+ }
233
+ function buildTraceInsightPrompt(input) {
234
+ const context = buildTraceInsightContext(input);
235
+ const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
236
+ return `Analyze this benchmark run and produce evidence-backed trace intelligence.
237
+
238
+ Audience:
239
+ - internal AI/product leadership
240
+ - possible customer-facing report for ${input.suite.name}
241
+
242
+ Investigation plan:
243
+ ${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
244
+
245
+ Analyst panel:
246
+ ${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
247
+
248
+ If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
249
+
250
+ Required output:
251
+ 1. Executive verdict: what this run proves and does not prove.
252
+ 2. The investigation questions you answered and the evidence used.
253
+ 3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
254
+ 4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
255
+ 5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
256
+ 6. What is safe for an external report versus what must stay internal.
257
+ 7. One rerun plan that would validate lift after optimization.
258
+
259
+ Budget:
260
+ - Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
261
+ - Prefer traces named in the failure summary over broad exploration.
262
+ - Do not do exhaustive trace sweeps.
263
+ - Return the final report as soon as the taxonomy and examples are supported.
264
+
265
+ Run summary:
266
+ ${JSON.stringify({
267
+ suite: input.suite.name,
268
+ scope: context.scope,
269
+ inferredKeywords: context.keywords,
270
+ agent: context.agent,
271
+ totals: context.totals,
272
+ findings: context.findings.map((finding) => ({
273
+ kind: finding.kind,
274
+ severity: finding.severity,
275
+ taskCount: finding.taskIds.length,
276
+ proposedFixClass: finding.proposedFixClass
277
+ })),
278
+ failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
279
+ task: task.id,
280
+ difficulty: task.difficulty,
281
+ outcome: task.outcome,
282
+ score: task.score,
283
+ gaps: task.gaps ?? []
284
+ }))
285
+ }, null, 2)}
286
+
287
+ Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
288
+ }
289
+ //#endregion
290
+ //#region src/trace-analyst/otlp-flatten.ts
291
+ const DEFAULT_KIND_MAP = {
292
+ 0: "SPAN_KIND_UNSPECIFIED",
293
+ 1: "SPAN_KIND_INTERNAL",
294
+ 2: "SPAN_KIND_SERVER",
295
+ 3: "SPAN_KIND_CLIENT",
296
+ 4: "SPAN_KIND_PRODUCER",
297
+ 5: "SPAN_KIND_CONSUMER"
298
+ };
299
+ const STATUS_MAP = {
300
+ 0: "STATUS_CODE_UNSET",
301
+ 1: "STATUS_CODE_OK",
302
+ 2: "STATUS_CODE_ERROR"
303
+ };
304
+ /** Unwrap an OTLP attribute-value union to a scalar. */
305
+ function attrValue(v) {
306
+ if (v.stringValue !== void 0) return v.stringValue;
307
+ if (v.intValue !== void 0) return Number(v.intValue);
308
+ if (v.doubleValue !== void 0) return v.doubleValue;
309
+ if (v.boolValue !== void 0) return v.boolValue;
310
+ return "";
311
+ }
312
+ function attrsToRecord(attrs) {
313
+ const out = {};
314
+ for (const a of attrs) out[a.key] = attrValue(a.value);
315
+ return out;
316
+ }
317
+ function nanoToIso(nano) {
318
+ const ms = Number(nano) / 1e6;
319
+ return Number.isFinite(ms) ? new Date(ms).toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
320
+ }
321
+ /** Mirror selected attributes into the OpenInference vocabulary in place. */
322
+ function applyOpenInference(attrs) {
323
+ if ("llm.model" in attrs && !("llm.model_name" in attrs)) attrs[LLM_MODEL_NAME] = attrs["llm.model"];
324
+ if ("llm.input_tokens" in attrs && !("llm.token_count.prompt" in attrs)) attrs[LLM_INPUT_TOKENS] = attrs["llm.input_tokens"];
325
+ if ("inference.llm.input_tokens" in attrs && !("llm.token_count.prompt" in attrs)) attrs[LLM_INPUT_TOKENS] = attrs["inference.llm.input_tokens"];
326
+ if ("llm.output_tokens" in attrs && !("llm.token_count.completion" in attrs)) attrs[LLM_OUTPUT_TOKENS] = attrs["llm.output_tokens"];
327
+ if ("inference.llm.output_tokens" in attrs && !("llm.token_count.completion" in attrs)) attrs[LLM_OUTPUT_TOKENS] = attrs["inference.llm.output_tokens"];
328
+ if ("tool.name" in attrs && !("inference.tool.name" in attrs)) attrs["inference.tool.name"] = attrs[TOOL_NAME];
329
+ if ("span.kind" in attrs && !("openinference.span.kind" in attrs)) attrs[OPENINFERENCE_SPAN_KIND] = String(attrs["span.kind"]).toUpperCase();
330
+ }
331
+ function flattenOtlpExportToNdjson(otlpExport, opts = {}) {
332
+ const vocab = opts.attributeVocabulary ?? "openinference";
333
+ const kindMap = {
334
+ ...DEFAULT_KIND_MAP,
335
+ ...opts.kindMap
336
+ };
337
+ const lines = [];
338
+ for (const rs of otlpExport.resourceSpans ?? []) {
339
+ const resource = { attributes: attrsToRecord(rs.resource?.attributes ?? []) };
340
+ for (const scope of rs.scopeSpans ?? []) for (const span of scope.spans ?? []) {
341
+ const attributes = attrsToRecord(span.attributes ?? []);
342
+ if (vocab === "openinference") applyOpenInference(attributes);
343
+ const line = {
344
+ trace_id: span.traceId,
345
+ span_id: span.spanId,
346
+ parent_span_id: span.parentSpanId ?? null,
347
+ name: span.name,
348
+ kind: kindMap[span.kind] ?? "SPAN_KIND_UNSPECIFIED",
349
+ start_time: nanoToIso(span.startTimeUnixNano),
350
+ end_time: nanoToIso(span.endTimeUnixNano),
351
+ status: {
352
+ code: STATUS_MAP[span.status?.code ?? 0] ?? "STATUS_CODE_UNSET",
353
+ ...span.status?.message ? { message: span.status.message } : {}
354
+ },
355
+ resource,
356
+ attributes
357
+ };
358
+ if (span.events && span.events.length > 0) line.events = span.events.map((e) => ({
359
+ name: e.name,
360
+ timeUnixNano: e.timeUnixNano,
361
+ ...e.attributes ? { attributes: attrsToRecord(e.attributes) } : {}
362
+ }));
363
+ lines.push(line);
364
+ }
365
+ }
366
+ return lines;
367
+ }
368
+ //#endregion
369
+ //#region src/trace-analyst/otlp-to-run-records.ts
370
+ /**
371
+ * `otlpToRunRecords` — fold an OTLP traces.jsonl (one OTLP span per line;
372
+ * the form AppWorld / HALO emit via their OpenInference OTLP exporter, the
373
+ * same shape `flattenOtlpExportToNdjson` produces) into validated
374
+ * `RunRecord[]` — one record per `trace_id` by default, or per caller-defined
375
+ * logical run when one task is fragmented across multiple OTLP traces.
376
+ *
377
+ * This is the offline ingestion primitive the AppWorld proposer bench and the
378
+ * hosted Intelligence product both stand on: traces in, paper-grade rows
379
+ * out, ready for `compareOptimizationMethods` / `analyzeRuns` / the promotion gate.
380
+ *
381
+ * Aggregation per trace:
382
+ * - tokenUsage: reconcile input, output, cache-read, and cache-write across
383
+ * nested model-call wrappers without double-counting parent aggregates.
384
+ * - costUsd: reconcile complete observed model-call cost when present; else priced via
385
+ * `opts.priceUsdPerToken` from the aggregated tokens; else `null` with a
386
+ * loud `raw.cost_unpriced = 1` marker.
387
+ * - task failure class and detail: read from process-root
388
+ * `tangle.task.failure_*` attributes; malformed or conflicting values throw.
389
+ * - terminalFailureReason: the failed root's normalized status message,
390
+ * when one unambiguous root supplies terminal failure evidence.
391
+ * - terminalOutcome: reduced from root-span status only. Child tool errors
392
+ * remain visible in `error_span_count` and `execution_error_count` without
393
+ * changing the run outcome. Root, guardrail, evaluator, propagated, and
394
+ * unknown errors retain separate counters.
395
+ * - model: the dominant LLM model in the trace (snapshot-padded to satisfy
396
+ * `validateRunRecord` when the trace's model is a bare alias).
397
+ * - outcome score: `opts.scoreForTrace` (AppWorld `world.evaluate()` →
398
+ * TGC/SGC) when supplied. Traces without an external task-quality signal
399
+ * remain unlabeled; execution errors never become a task score.
400
+ * - prompt / completion: carried into `raw` as token-count signals and,
401
+ * when the first/last LLM span exposes `input.value` / `output.value`,
402
+ * the verbatim text is preserved on the optional `promptText` /
403
+ * `completionText` of the returned `OtlpTraceRunRecord`.
404
+ *
405
+ * Fail-loud: an OTLP file with zero valid spans throws. A trace with no
406
+ * spans is impossible (a trace exists only because a span referenced it).
407
+ * `validateRunRecord` runs on every row — a malformed projection throws
408
+ * rather than silently producing a half-record.
409
+ */
410
+ /**
411
+ * Parse + aggregate an OTLP traces.jsonl string into validated
412
+ * `RunRecord[]` (one per trace). Use {@link otlpToTraceRunRecords} when you
413
+ * also want the verbatim prompt/completion text alongside each record.
414
+ */
415
+ function otlpToRunRecords(otlpJsonl, opts) {
416
+ return otlpToTraceRunRecords(otlpJsonl, opts).map((r) => r.record);
417
+ }
418
+ /**
419
+ * Aggregate already-parsed OTLP flat rows without serializing them back to
420
+ * JSONL. This is the in-memory counterpart to {@link otlpToRunRecords}; both
421
+ * paths share projection, reconciliation, validation, and ordering.
422
+ */
423
+ function otlpRowsToRunRecords(rows, opts) {
424
+ return otlpRowsToTraceRunRecords(rows, opts).map((row) => row.record);
425
+ }
426
+ /** As {@link otlpToRunRecords} but returns the prompt/completion text too. */
427
+ function otlpToTraceRunRecords(otlpJsonl, opts) {
428
+ return traceRunRecordsFromSpans(groupSpansByLogicalRun(groupJsonlSpansByTrace(otlpJsonl), opts.logicalRunIdForTrace), opts);
429
+ }
430
+ /** Parsed-row counterpart to {@link otlpToTraceRunRecords}. */
431
+ function otlpRowsToTraceRunRecords(rows, opts) {
432
+ return traceRunRecordsFromSpans(groupSpansByLogicalRun(groupRowsByTrace(rows), opts.logicalRunIdForTrace), opts);
433
+ }
434
+ function traceRunRecordsFromSpans(byTrace, opts) {
435
+ const splitTag = opts.splitTag ?? "holdout";
436
+ const commitSha = opts.commitSha ?? "unknown";
437
+ const promptHash = opts.promptHash ?? "unknown";
438
+ const configHash = opts.configHash ?? "unknown";
439
+ const seed = opts.seed ?? 0;
440
+ const fallbackModel = opts.fallbackModel ?? "unknown@otlp";
441
+ if (byTrace.size === 0) throw new Error("otlpToRunRecords: OTLP input produced zero valid spans — every row was empty, malformed, or missing trace_id/span_id");
442
+ const traceIds = [...byTrace.keys()].sort();
443
+ const out = [];
444
+ for (const traceId of traceIds) {
445
+ const spans = byTrace.get(traceId);
446
+ const agg = aggregateTrace(traceId, spans, fallbackModel);
447
+ const score = resolveScore(opts, traceId, agg);
448
+ const { costUsd, costProvenance } = resolveCost(opts, agg);
449
+ const raw = {
450
+ source_trace_count: agg.sourceTraceCount,
451
+ span_count: agg.spanCount,
452
+ llm_span_count: agg.llmSpanCount,
453
+ tool_span_count: agg.toolSpanCount,
454
+ agent_span_count: agg.agentSpanCount,
455
+ error_span_count: agg.errorSpanCount,
456
+ execution_error_count: agg.executionErrorCount,
457
+ process_error_count: agg.processErrorCount,
458
+ guardrail_error_count: agg.guardrailErrorCount,
459
+ judge_error_count: agg.judgeErrorCount,
460
+ propagated_error_count: agg.propagatedErrorCount,
461
+ unclassified_error_count: agg.unclassifiedErrorCount,
462
+ prompt_tokens: agg.tokenUsage.input,
463
+ completion_tokens: agg.tokenUsage.output
464
+ };
465
+ if (agg.tokenUsage.reasoning !== void 0) raw.reasoning_tokens = agg.tokenUsage.reasoning;
466
+ if (agg.tokenUsage.cached !== void 0) raw.cached_tokens = agg.tokenUsage.cached;
467
+ if (agg.tokenUsage.cacheWrite !== void 0) raw.cache_write_tokens = agg.tokenUsage.cacheWrite;
468
+ if (agg.costMeasurement.value !== void 0 && !agg.costMeasurement.complete) raw.partial_observed_cost_usd = agg.costMeasurement.value;
469
+ recordAggregateMeasurements(raw, agg.aggregateMeasurement);
470
+ if (costProvenance.kind === "uncaptured") raw.cost_unpriced = 1;
471
+ const outcome = { raw };
472
+ if (score !== void 0) if (splitTag === "holdout") outcome.holdoutScore = score;
473
+ else outcome.searchScore = score;
474
+ const { promptText, completionText } = extractPromptCompletion(spans, agg.callSpanIds);
475
+ const judgeMetadata = opts.judgeMetadataForTrace?.(traceId);
476
+ const taskFailure = readTaskFailureLabels(spans.filter((span) => span.parent_span_id === null && isTerminalRootCandidate(span)), `otlpToRunRecords: run '${traceId}'`);
477
+ const record = validateRunRecord({
478
+ runId: `otlp:${opts.experimentId}:${opts.candidateId}:${traceId}`,
479
+ experimentId: opts.experimentId,
480
+ candidateId: opts.candidateId,
481
+ seed,
482
+ model: ensureSnapshot(agg.model, fallbackModel),
483
+ promptHash,
484
+ configHash,
485
+ commitSha,
486
+ wallMs: agg.wallMs,
487
+ costUsd,
488
+ costProvenance,
489
+ tokenUsage: agg.tokenUsage,
490
+ terminalOutcome: agg.terminalOutcome,
491
+ ...agg.terminalFailureMessage ? { terminalFailureReason: agg.terminalFailureMessage } : {},
492
+ ...judgeMetadata ? { judgeMetadata } : {},
493
+ outcome,
494
+ ...taskFailure,
495
+ splitTag,
496
+ scenarioId: traceId
497
+ });
498
+ out.push({
499
+ record,
500
+ ...promptText !== void 0 ? { promptText } : {},
501
+ ...completionText !== void 0 ? { completionText } : {}
502
+ });
503
+ }
504
+ return out;
505
+ }
506
+ function* yieldJsonlRows(otlpJsonl) {
507
+ for (const line of otlpJsonl.split("\n")) {
508
+ const trimmed = line.trim();
509
+ if (trimmed.length === 0) continue;
510
+ let parsed;
511
+ try {
512
+ parsed = JSON.parse(trimmed);
513
+ } catch {
514
+ continue;
515
+ }
516
+ if (parsed && typeof parsed === "object") yield parsed;
517
+ }
518
+ }
519
+ function groupJsonlSpansByTrace(otlpJsonl) {
520
+ return groupRowsByTrace(yieldJsonlRows(otlpJsonl));
521
+ }
522
+ function groupRowsByTrace(rows) {
523
+ const byTrace = /* @__PURE__ */ new Map();
524
+ for (const row of rows) {
525
+ if (!row || typeof row !== "object") continue;
526
+ const span = projectOtlpFlatLine(row);
527
+ if (!span) continue;
528
+ const arr = byTrace.get(span.trace_id);
529
+ if (arr) arr.push(span);
530
+ else byTrace.set(span.trace_id, [span]);
531
+ }
532
+ return byTrace;
533
+ }
534
+ function groupSpansByLogicalRun(byTrace, logicalRunIdForTrace) {
535
+ if (!logicalRunIdForTrace) return byTrace;
536
+ const byRun = /* @__PURE__ */ new Map();
537
+ for (const [traceId, spans] of byTrace) {
538
+ const suppliedRunId = logicalRunIdForTrace(traceId);
539
+ if (typeof suppliedRunId !== "string" || suppliedRunId.trim().length === 0) throw new Error(`otlpToRunRecords: logicalRunIdForTrace('${traceId}') returned an empty run id`);
540
+ const runId = suppliedRunId.trim();
541
+ const target = byRun.get(runId) ?? [];
542
+ for (const span of spans) target.push({
543
+ ...span,
544
+ span_id: qualifySpanId(traceId, span.span_id),
545
+ parent_span_id: span.parent_span_id ? qualifySpanId(traceId, span.parent_span_id) : null
546
+ });
547
+ byRun.set(runId, target);
548
+ }
549
+ return byRun;
550
+ }
551
+ function qualifySpanId(traceId, spanId) {
552
+ const prefix = `${traceId}:`;
553
+ return spanId.startsWith(prefix) ? spanId : `${prefix}${spanId}`;
554
+ }
555
+ function aggregateTrace(traceId, spans, fallbackModel) {
556
+ const ordered = [...spans].sort((a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id));
557
+ const measurements = summarizeExecutionMeasurements(ordered.map((span) => ({
558
+ id: span.span_id,
559
+ ...span.parent_span_id ? { parentId: span.parent_span_id } : {},
560
+ attributes: span.attributes,
561
+ modelCall: isOtlpModelCall({
562
+ kind: span.kind,
563
+ name: span.name,
564
+ attributes: span.attributes
565
+ }),
566
+ aggregate: span.kind !== "LLM" && span.kind !== "UNKNOWN"
567
+ })));
568
+ let toolSpanCount = 0;
569
+ let agentSpanCount = 0;
570
+ let firstErrorMessage;
571
+ const modelVotes = /* @__PURE__ */ new Map();
572
+ let earliest = ordered[0]?.start_time ?? "";
573
+ let latest = ordered[0]?.end_time ?? "";
574
+ for (const s of ordered) {
575
+ if (s.start_time && (!earliest || compareSpanTime(s.start_time, earliest) < 0)) earliest = s.start_time;
576
+ if (s.end_time && (!latest || compareSpanTime(s.end_time, latest) > 0)) latest = s.end_time;
577
+ if (s.kind === "TOOL") toolSpanCount += 1;
578
+ else if (s.kind === "AGENT") agentSpanCount += 1;
579
+ if (s.status === "ERROR") {
580
+ if (firstErrorMessage === void 0) firstErrorMessage = (s.status_message ?? `${s.name} — STATUS_CODE_ERROR`).slice(0, 500);
581
+ }
582
+ }
583
+ const callSpanIds = new Set(measurements.callSpanIds);
584
+ for (const span of ordered) {
585
+ if (!callSpanIds.has(span.span_id)) continue;
586
+ const model = firstStringAttr(span.attributes, LLM_MODEL_ATTR_KEYS) ?? span.model_name;
587
+ if (model) modelVotes.set(model, (modelVotes.get(model) ?? 0) + 1);
588
+ }
589
+ const model = topVote(modelVotes) ?? firstModelAttr(ordered) ?? fallbackModel;
590
+ let wallMs = 0;
591
+ const a = spanEpochMillis(earliest);
592
+ const b = spanEpochMillis(latest);
593
+ if (a !== null && b !== null) wallMs = Math.max(0, b - a);
594
+ const sourceTraceIds = [...new Set(spans.map((span) => span.trace_id))].sort();
595
+ const terminal = terminalEvidenceFromRoots(ordered);
596
+ const errorSummary = summarizeTraceErrors(ordered.map((span) => ({
597
+ id: span.span_id,
598
+ ...span.parent_span_id ? { parentId: span.parent_span_id } : {},
599
+ role: errorRoleForProjectedSpan(span),
600
+ error: span.status === "ERROR",
601
+ processRoot: span.parent_span_id === null && isTerminalRootCandidate(span)
602
+ })));
603
+ return {
604
+ traceId,
605
+ sourceTraceCount: sourceTraceIds.length,
606
+ sourceTraceIds,
607
+ spanCount: spans.length,
608
+ llmSpanCount: measurements.modelCallCount,
609
+ toolSpanCount,
610
+ agentSpanCount,
611
+ errorSpanCount: errorSummary.total,
612
+ executionErrorCount: errorSummary.execution,
613
+ processErrorCount: errorSummary.process,
614
+ guardrailErrorCount: errorSummary.guardrail,
615
+ judgeErrorCount: errorSummary.evaluation,
616
+ propagatedErrorCount: errorSummary.propagated,
617
+ unclassifiedErrorCount: errorSummary.unclassified,
618
+ tokenUsage: measurements.tokenUsage,
619
+ firstErrorMessage,
620
+ model,
621
+ startTime: earliest,
622
+ endTime: latest,
623
+ wallMs,
624
+ terminalOutcome: terminal.outcome,
625
+ ...terminal.failureMessage ? { terminalFailureMessage: terminal.failureMessage } : {},
626
+ callSpanIds: measurements.callSpanIds,
627
+ costMeasurement: measurements.cost,
628
+ ...measurements.aggregate ? { aggregateMeasurement: measurements.aggregate } : {}
629
+ };
630
+ }
631
+ function terminalEvidenceFromRoots(spans) {
632
+ const roots = spans.filter((span) => span.parent_span_id === null && isTerminalRootCandidate(span));
633
+ if (roots.length !== 1) return { outcome: "unknown" };
634
+ const root = roots[0];
635
+ if (root.status === "ERROR") return {
636
+ outcome: "failed",
637
+ failureMessage: (root.status_message ?? `${root.name} — STATUS_CODE_ERROR`).slice(0, 500)
638
+ };
639
+ if (root.status === "OK") return { outcome: "succeeded" };
640
+ return { outcome: "unknown" };
641
+ }
642
+ function isTerminalRootCandidate(span) {
643
+ const role = errorRoleForProjectedSpan(span);
644
+ return role !== "LLM" && role !== "TOOL" && role !== "EVALUATOR" && role !== "GUARDRAIL";
645
+ }
646
+ function errorRoleForProjectedSpan(span) {
647
+ return classifyOtlpSpanRole({
648
+ kind: span.kind,
649
+ name: span.name,
650
+ attributes: span.attributes
651
+ });
652
+ }
653
+ function resolveScore(opts, traceId, agg) {
654
+ const supplied = opts.scoreForTrace?.(traceId, agg);
655
+ if (supplied !== void 0) {
656
+ if (!Number.isFinite(supplied)) throw new Error(`otlpToRunRecords: scoreForTrace('${traceId}') returned non-finite ${supplied}`);
657
+ return supplied;
658
+ }
659
+ }
660
+ function resolveCost(opts, agg) {
661
+ const observedCost = agg.costMeasurement;
662
+ if (observedCost.complete && observedCost.value !== void 0) return {
663
+ costUsd: observedCost.value,
664
+ costProvenance: {
665
+ kind: "observed",
666
+ usd: observedCost.value
667
+ }
668
+ };
669
+ if (agg.aggregateMeasurement?.costUsd !== void 0) return {
670
+ costUsd: agg.aggregateMeasurement.costUsd,
671
+ costProvenance: {
672
+ kind: "observed",
673
+ usd: agg.aggregateMeasurement.costUsd
674
+ }
675
+ };
676
+ if (opts.priceUsdPerToken !== void 0) {
677
+ const costUsd = (agg.tokenUsage.input + agg.tokenUsage.output) * opts.priceUsdPerToken;
678
+ return {
679
+ costUsd,
680
+ costProvenance: {
681
+ kind: "estimated",
682
+ usd: costUsd
683
+ }
684
+ };
685
+ }
686
+ return {
687
+ costUsd: null,
688
+ costProvenance: {
689
+ kind: "uncaptured",
690
+ usd: null
691
+ }
692
+ };
693
+ }
694
+ function extractPromptCompletion(spans, callSpanIds) {
695
+ const callIds = new Set(callSpanIds);
696
+ const measuredCalls = spans.filter((span) => callIds.has(span.span_id));
697
+ const llm = (measuredCalls.length > 0 ? measuredCalls : spans.filter((s) => s.kind === "LLM")).sort((a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id));
698
+ if (llm.length === 0) return {};
699
+ const promptText = firstStringAttr(llm[0].attributes, [
700
+ "input.value",
701
+ "llm.input_messages",
702
+ "gen_ai.prompt"
703
+ ]) ?? void 0;
704
+ const last = llm[llm.length - 1];
705
+ const completionText = firstStringAttr(last.attributes, [
706
+ "output.value",
707
+ "llm.output_messages",
708
+ "gen_ai.completion"
709
+ ]) ?? void 0;
710
+ return {
711
+ ...promptText !== void 0 ? { promptText } : {},
712
+ ...completionText !== void 0 ? { completionText } : {}
713
+ };
714
+ }
715
+ function topVote(votes) {
716
+ let best = null;
717
+ let bestN = 0;
718
+ for (const [k, n] of votes) if (n > bestN || n === bestN && best !== null && k < best) {
719
+ best = k;
720
+ bestN = n;
721
+ }
722
+ return best;
723
+ }
724
+ function firstModelAttr(spans) {
725
+ for (const s of spans) {
726
+ const m = firstStringAttr(s.attributes, LLM_MODEL_ATTR_KEYS) ?? s.model_name;
727
+ if (m) return m;
728
+ }
729
+ return null;
730
+ }
731
+ /**
732
+ * `validateRunRecord` rejects bare model aliases (`gpt-4o`) that remap
733
+ * silently. AppWorld/HALO traces frequently carry such bare ids (or a null
734
+ * model). When the model already encodes a snapshot we keep it; otherwise we
735
+ * append the fallback snapshot token so the row is admissible without lying
736
+ * about the model — the bare base name is preserved verbatim before `@`.
737
+ */
738
+ function ensureSnapshot(model, fallbackModel) {
739
+ if (modelHasSnapshot(model)) return model;
740
+ return `${model}${fallbackModel.includes("@") ? fallbackModel.slice(fallbackModel.indexOf("@")) : "@otlp"}`;
741
+ }
742
+ function modelHasSnapshot(model) {
743
+ if (model.includes("@")) return true;
744
+ if (/-\d{8}$/.test(model)) return true;
745
+ if (/-\d{4}-\d{2}-\d{2}$/.test(model)) return true;
746
+ if (/:date-/.test(model)) return true;
747
+ return false;
748
+ }
749
+ //#endregion
750
+ //#region src/trace/capture-fetch.ts
751
+ /**
752
+ * Wrap a provider `fetch` and record request, response, and error events.
753
+ *
754
+ * The returned value is a plain `typeof fetch`. Capture is best-effort by
755
+ * default; set `failClosed` when telemetry loss must stop the provider call.
756
+ */
757
+ const DEFAULT_BODY_CAP = 2 * 1024 * 1024;
758
+ function headersToRecord(headers) {
759
+ if (!headers) return void 0;
760
+ const out = {};
761
+ headers.forEach((value, key) => {
762
+ out[key.toLowerCase()] = value;
763
+ });
764
+ return Object.keys(out).length > 0 ? out : void 0;
765
+ }
766
+ function parseMaybeJson(text) {
767
+ if (text.length === 0) return void 0;
768
+ try {
769
+ return JSON.parse(text);
770
+ } catch {
771
+ return text;
772
+ }
773
+ }
774
+ /** Best-effort request-body read across the `fetch` input forms. */
775
+ async function readRequestBody(input, init) {
776
+ if (typeof init?.body === "string") return parseMaybeJson(init.body);
777
+ if (init?.body != null) return void 0;
778
+ if (input instanceof Request) try {
779
+ return parseMaybeJson(await input.clone().text());
780
+ } catch {
781
+ return;
782
+ }
783
+ }
784
+ function endpointFromUrl(url, baseUrl) {
785
+ const normalisedBase = baseUrl.replace(/\/+$/, "");
786
+ if (url.startsWith(normalisedBase)) return url.slice(normalisedBase.length) || "/";
787
+ try {
788
+ return new URL(url).pathname;
789
+ } catch {
790
+ return url;
791
+ }
792
+ }
793
+ function captureFetchToRawSink(fetch, sink, ctx, opts = {}) {
794
+ const provider = ctx.provider ?? providerFromBaseUrl(ctx.baseUrl);
795
+ const redactor = opts.redactor ?? defaultProviderRedactor;
796
+ const bodyCap = opts.responseBodyByteCap ?? DEFAULT_BODY_CAP;
797
+ let warned = false;
798
+ const baseEvent = (direction, endpoint) => ({
799
+ eventId: crypto.randomUUID(),
800
+ runId: ctx.runId,
801
+ spanId: ctx.spanId,
802
+ provider,
803
+ model: ctx.model,
804
+ endpoint,
805
+ baseUrl: ctx.baseUrl,
806
+ attemptIndex: 0,
807
+ direction,
808
+ timestamp: Date.now(),
809
+ redactedFields: []
810
+ });
811
+ const record = async (event) => {
812
+ try {
813
+ await sink.record(redactor(event));
814
+ } catch (err) {
815
+ if (opts.failClosed) throw err;
816
+ if (!warned) {
817
+ warned = true;
818
+ console.warn(`captureFetchToRawSink: sink.record failed (capture is best-effort) — ${err instanceof Error ? err.message : String(err)}`);
819
+ }
820
+ }
821
+ };
822
+ return async (input, init) => {
823
+ const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
824
+ const method = (init?.method ?? (input instanceof Request ? input.method : "GET")).toUpperCase();
825
+ const endpoint = endpointFromUrl(url, ctx.baseUrl);
826
+ const reqHeaders = new Headers(init?.headers ?? (input instanceof Request ? input.headers : void 0));
827
+ await record({
828
+ ...baseEvent("request", endpoint),
829
+ requestHeaders: {
830
+ ...headersToRecord(reqHeaders),
831
+ "x-http-method": method
832
+ },
833
+ requestBody: await readRequestBody(input, init)
834
+ });
835
+ const start = Date.now();
836
+ let response;
837
+ try {
838
+ response = await fetch(input, init);
839
+ } catch (err) {
840
+ await record({
841
+ ...baseEvent("error", endpoint),
842
+ durationMs: Date.now() - start,
843
+ errorMessage: err instanceof Error ? err.message : String(err)
844
+ });
845
+ throw err;
846
+ }
847
+ let responseBody;
848
+ let rawText;
849
+ const redactedFields = [];
850
+ try {
851
+ rawText = await response.clone().text();
852
+ if (rawText.length > bodyCap) {
853
+ responseBody = rawText.slice(0, bodyCap);
854
+ redactedFields.push("body_truncated");
855
+ } else responseBody = parseMaybeJson(rawText);
856
+ } catch {
857
+ responseBody = void 0;
858
+ }
859
+ if (opts.onUsage && rawText !== void 0) try {
860
+ const parsedForUsage = parseMaybeJson(rawText);
861
+ const usage = extractUsage(parsedForUsage) ?? (typeof parsedForUsage === "string" ? extractUsageFromSse(rawText, { mode: opts.sseUsageMode }) : null);
862
+ if (usage) opts.onUsage(usage, ctx);
863
+ } catch (err) {
864
+ if (opts.failClosed) throw err;
865
+ }
866
+ await record({
867
+ ...baseEvent("response", endpoint),
868
+ durationMs: Date.now() - start,
869
+ statusCode: response.status,
870
+ responseHeaders: headersToRecord(response.headers),
871
+ responseBody,
872
+ redactedFields
873
+ });
874
+ return response;
875
+ };
876
+ }
877
+ //#endregion
878
+ //#region src/trace/otel.ts
879
+ /**
880
+ * OpenTelemetry JSON export — maps TraceSchema v1 to OTLP/JSON so
881
+ * traces render natively in Jaeger / Honeycomb / Langfuse / Grafana.
882
+ *
883
+ * Wire format only. We do NOT depend on the @opentelemetry SDK — that
884
+ * would drag in polyfills incompatible with Workers/Edge. Consumers
885
+ * push the JSON to their collector of choice via HTTP.
886
+ *
887
+ * Reference: OTLP 1.3.2 (ResourceSpans / ScopeSpans / Span).
888
+ */
889
+ const OTEL_AGENT_EVAL_SCOPE = {
890
+ name: "@tangle-network/agent-eval",
891
+ version: "0.3.0"
892
+ };
893
+ /** Export a single run's spans + events in OTLP/JSON. */
894
+ async function exportRunAsOtlp(store, runId, resourceAttrs = {}) {
895
+ const run = await store.getRun(runId);
896
+ if (!run) throw new Error(`run ${runId} not found`);
897
+ const spans = await store.spans({ runId });
898
+ const events = await store.events({ runId });
899
+ const eventsBySpan = /* @__PURE__ */ new Map();
900
+ for (const e of events) {
901
+ if (!e.spanId) continue;
902
+ const arr = eventsBySpan.get(e.spanId) ?? [];
903
+ arr.push(e);
904
+ eventsBySpan.set(e.spanId, arr);
905
+ }
906
+ const traceId = runToTraceId(run);
907
+ const otlpSpans = spans.map((s) => spanToOtlp(s, traceId, eventsBySpan.get(s.spanId) ?? []));
908
+ return { resourceSpans: [{
909
+ resource: { attributes: toAttributes$1({
910
+ "service.name": "agent-eval",
911
+ "run.id": run.runId,
912
+ "run.scenario_id": run.scenarioId,
913
+ "run.variant_id": run.variantId ?? "",
914
+ "run.dataset_version": run.datasetVersion ?? "",
915
+ "run.code_sha": run.codeSha ?? "",
916
+ "run.model_fingerprint": run.modelFingerprint ?? "",
917
+ ...resourceAttrs
918
+ }) },
919
+ scopeSpans: [{
920
+ scope: OTEL_AGENT_EVAL_SCOPE,
921
+ spans: otlpSpans
922
+ }]
923
+ }] };
924
+ }
925
+ function spanToOtlp(span, traceId, events) {
926
+ const endedAt = span.endedAt ?? span.startedAt;
927
+ return {
928
+ traceId,
929
+ spanId: padSpanId$2(span.spanId),
930
+ parentSpanId: span.parentSpanId ? padSpanId$2(span.parentSpanId) : void 0,
931
+ name: span.name,
932
+ kind: 1,
933
+ startTimeUnixNano: msToNs$1(span.startedAt),
934
+ endTimeUnixNano: msToNs$1(endedAt),
935
+ attributes: toAttributes$1(flattenSpanAttributes(span)),
936
+ events: events.map((e) => ({
937
+ timeUnixNano: msToNs$1(e.timestamp),
938
+ name: e.kind,
939
+ attributes: toAttributes$1(flattenPayload(e.payload))
940
+ })),
941
+ status: span.status === "error" ? {
942
+ code: 2,
943
+ message: span.error
944
+ } : { code: 1 }
945
+ };
946
+ }
947
+ function flattenSpanAttributes(span) {
948
+ const base = {};
949
+ if (span.attributes) {
950
+ for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") base[k] = v;
951
+ }
952
+ base[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
953
+ if (span.kind === "llm") applyLlmSpanOtlpAttributes(base, span);
954
+ else if (span.kind === "tool") applyToolSpanOtlpAttributes(base, span);
955
+ else if (span.kind === "retrieval") {
956
+ base["retrieval.query"] = span.query;
957
+ base["retrieval.hits"] = span.hits.length;
958
+ } else if (span.kind === "judge") {
959
+ base["judge.id"] = span.judgeId;
960
+ base["judge.dimension"] = span.dimension;
961
+ base["judge.score"] = span.score;
962
+ base["judge.target_span_id"] = span.targetSpanId;
963
+ } else if (span.kind === "sandbox") {
964
+ if (span.image) base["sandbox.image"] = span.image;
965
+ if (span.exitCode !== void 0) base["sandbox.exit_code"] = span.exitCode;
966
+ if (span.testsPassed !== void 0) base["sandbox.tests_passed"] = span.testsPassed;
967
+ if (span.testsTotal !== void 0) base["sandbox.tests_total"] = span.testsTotal;
968
+ }
969
+ return base;
970
+ }
971
+ function flattenPayload(payload) {
972
+ const out = {};
973
+ for (const [k, v] of Object.entries(payload)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") out[k] = v;
974
+ else out[k] = JSON.stringify(v);
975
+ return out;
976
+ }
977
+ function toAttributes$1(record) {
978
+ return Object.entries(record).map(([key, value]) => ({
979
+ key,
980
+ value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
981
+ }));
982
+ }
983
+ function msToNs$1(ms) {
984
+ return (BigInt(Math.floor(ms)) * 1000000n).toString();
985
+ }
986
+ function padSpanId$2(id) {
987
+ return id.replace(/-/g, "").slice(0, 16).padEnd(16, "0");
988
+ }
989
+ function runToTraceId(run) {
990
+ return run.runId.replace(/-/g, "").slice(0, 32).padEnd(32, "0");
991
+ }
992
+ //#endregion
993
+ //#region src/trace/otel-bridge.ts
994
+ /**
995
+ * Create a RunCompleteHook that exports all spans from the completed run
996
+ * to the OTEL exporter, then flushes.
997
+ */
998
+ function otelRunCompleteHook(exporter) {
999
+ return async (ctx) => {
1000
+ const spans = await ctx.store.spans({ runId: ctx.runId });
1001
+ for (const span of spans) if (span.endedAt) exporter.exportSpan(storeSpanToExportable(span, ctx.runId));
1002
+ await exporter.flush();
1003
+ };
1004
+ }
1005
+ /**
1006
+ * Create an auto-exporting TraceStore wrapper that intercepts updateSpan
1007
+ * calls. When a span gets an endedAt, it's exported immediately. This
1008
+ * gives real-time streaming instead of batch-at-end.
1009
+ *
1010
+ * This is the preferred integration path: wrap the store before
1011
+ * constructing the TraceEmitter.
1012
+ */
1013
+ function createOtelTracingStore(inner, exporter, traceId) {
1014
+ return {
1015
+ async appendRun(run) {
1016
+ return inner.appendRun(run);
1017
+ },
1018
+ async updateRun(runId, patch) {
1019
+ return inner.updateRun(runId, patch);
1020
+ },
1021
+ async appendSpan(span) {
1022
+ if (span.endedAt) exporter.exportSpan(storeSpanToExportable(span, traceId));
1023
+ return inner.appendSpan(span);
1024
+ },
1025
+ async updateSpan(spanId, patch) {
1026
+ await inner.updateSpan(spanId, patch);
1027
+ if (patch.endedAt) {
1028
+ const found = (await inner.spans({ runId: traceId })).find((s) => s.spanId === spanId);
1029
+ if (found) exporter.exportSpan(storeSpanToExportable(found, traceId));
1030
+ }
1031
+ },
1032
+ async appendEvent(event) {
1033
+ return inner.appendEvent(event);
1034
+ },
1035
+ async appendBudgetEntry(entry) {
1036
+ return inner.appendBudgetEntry(entry);
1037
+ },
1038
+ async appendArtifact(artifact) {
1039
+ return inner.appendArtifact(artifact);
1040
+ },
1041
+ getRun: inner.getRun.bind(inner),
1042
+ listRuns: inner.listRuns.bind(inner),
1043
+ spans: inner.spans.bind(inner),
1044
+ events: inner.events.bind(inner),
1045
+ budget: inner.budget.bind(inner),
1046
+ artifacts: inner.artifacts.bind(inner)
1047
+ };
1048
+ }
1049
+ function storeSpanToExportable(span, traceId) {
1050
+ const llm = span.kind === "llm" ? span : void 0;
1051
+ const tool = span.kind === "tool" ? span : void 0;
1052
+ return {
1053
+ traceId,
1054
+ spanId: span.spanId,
1055
+ parentSpanId: span.parentSpanId,
1056
+ name: span.name,
1057
+ kind: span.kind,
1058
+ startedAt: span.startedAt,
1059
+ endedAt: span.endedAt,
1060
+ status: span.status,
1061
+ error: span.error,
1062
+ model: llm?.model,
1063
+ inputTokens: llm?.inputTokens,
1064
+ outputTokens: llm?.outputTokens,
1065
+ reasoningTokens: llm?.reasoningTokens,
1066
+ cachedTokens: llm?.cachedTokens,
1067
+ cacheWriteTokens: llm?.cacheWriteTokens,
1068
+ costUsd: llm?.costUsd,
1069
+ tool: tool ? {
1070
+ toolName: tool.toolName,
1071
+ args: tool.args,
1072
+ argsCaptured: tool.argsCaptured,
1073
+ result: tool.result,
1074
+ latencyMs: tool.latencyMs
1075
+ } : void 0,
1076
+ attributes: span.attributes
1077
+ };
1078
+ }
1079
+ //#endregion
1080
+ //#region src/trace/otel-export.ts
1081
+ /**
1082
+ * OTEL span exporter — streams spans to an OTLP/HTTP collector.
1083
+ *
1084
+ * Reads OTEL_EXPORTER_OTLP_ENDPOINT + OTEL_EXPORTER_OTLP_HEADERS from env
1085
+ * when no explicit config is given. Batches spans and flushes periodically
1086
+ * or when the batch fills. No @opentelemetry SDK dependency — minimal
1087
+ * OTLP/JSON serializer (~120 LOC) using the existing otel.ts helpers.
1088
+ */
1089
+ /**
1090
+ * Create an OTEL exporter. Returns undefined when no endpoint is configured
1091
+ * (neither via config nor env) — callers should check before attaching.
1092
+ */
1093
+ function createOtelExporter(config) {
1094
+ const resolvedEndpoint = config?.endpoint ?? (typeof process !== "undefined" ? process.env.OTEL_EXPORTER_OTLP_ENDPOINT : void 0);
1095
+ if (!resolvedEndpoint) return void 0;
1096
+ const endpoint = resolvedEndpoint;
1097
+ const headers = config?.headers ?? parseHeadersFromEnv();
1098
+ const batchSize = config?.batchSize ?? 64;
1099
+ const flushIntervalMs = config?.flushIntervalMs ?? 5e3;
1100
+ const serviceName = config?.serviceName ?? "agent-eval";
1101
+ const resourceAttrs = config?.resourceAttributes ?? {};
1102
+ const pending = [];
1103
+ let timer;
1104
+ let stopped = false;
1105
+ const exporter = {
1106
+ exportSpan(span) {
1107
+ if (stopped) return;
1108
+ pending.push(toOtlpSpan(span));
1109
+ if (pending.length >= batchSize) doFlush();
1110
+ },
1111
+ async flush() {
1112
+ await doFlush();
1113
+ },
1114
+ async shutdown() {
1115
+ stopped = true;
1116
+ if (timer !== void 0) {
1117
+ clearInterval(timer);
1118
+ timer = void 0;
1119
+ }
1120
+ await doFlush();
1121
+ }
1122
+ };
1123
+ timer = setInterval(() => {
1124
+ if (pending.length > 0) doFlush();
1125
+ }, flushIntervalMs);
1126
+ if (typeof timer === "object" && "unref" in timer) timer.unref();
1127
+ async function doFlush() {
1128
+ if (pending.length === 0) return;
1129
+ const batch = pending.splice(0);
1130
+ const body = { resourceSpans: [{
1131
+ resource: { attributes: toAttributes({
1132
+ "service.name": serviceName,
1133
+ ...resourceAttrs
1134
+ }) },
1135
+ scopeSpans: [{
1136
+ scope: OTEL_AGENT_EVAL_SCOPE,
1137
+ spans: batch
1138
+ }]
1139
+ }] };
1140
+ const url = `${endpoint.replace(/\/+$/, "")}/v1/traces`;
1141
+ try {
1142
+ await fetch(url, {
1143
+ method: "POST",
1144
+ headers: {
1145
+ "content-type": "application/json",
1146
+ ...headers
1147
+ },
1148
+ body: JSON.stringify(body)
1149
+ });
1150
+ } catch {}
1151
+ }
1152
+ return exporter;
1153
+ }
1154
+ function parseHeadersFromEnv() {
1155
+ if (typeof process === "undefined") return {};
1156
+ const raw = process.env.OTEL_EXPORTER_OTLP_HEADERS;
1157
+ if (!raw) return {};
1158
+ const out = {};
1159
+ for (const pair of raw.split(",")) {
1160
+ const eq = pair.indexOf("=");
1161
+ if (eq < 0) continue;
1162
+ const key = pair.slice(0, eq).trim();
1163
+ const value = pair.slice(eq + 1).trim();
1164
+ if (key) out[key] = value;
1165
+ }
1166
+ return out;
1167
+ }
1168
+ function toOtlpSpan(span) {
1169
+ const endedAt = span.endedAt ?? span.startedAt;
1170
+ const attrs = {};
1171
+ if (span.attributes) {
1172
+ for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") attrs[k] = v;
1173
+ }
1174
+ attrs[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
1175
+ applyLlmSpanOtlpAttributes(attrs, span);
1176
+ if (span.tool) applyToolSpanOtlpAttributes(attrs, span.tool);
1177
+ return {
1178
+ traceId: padTraceId$1(span.traceId),
1179
+ spanId: padSpanId$1(span.spanId),
1180
+ parentSpanId: span.parentSpanId ? padSpanId$1(span.parentSpanId) : void 0,
1181
+ name: span.name,
1182
+ kind: 1,
1183
+ startTimeUnixNano: msToNs(span.startedAt),
1184
+ endTimeUnixNano: msToNs(endedAt),
1185
+ attributes: toAttributes(attrs),
1186
+ status: span.status === "error" ? {
1187
+ code: 2,
1188
+ message: span.error
1189
+ } : { code: 1 }
1190
+ };
1191
+ }
1192
+ function toAttributes(record) {
1193
+ return Object.entries(record).map(([key, value]) => ({
1194
+ key,
1195
+ value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
1196
+ }));
1197
+ }
1198
+ function msToNs(ms) {
1199
+ return (BigInt(Math.floor(ms)) * 1000000n).toString();
1200
+ }
1201
+ function padSpanId$1(id) {
1202
+ return id.replace(/-/g, "").slice(0, 16).padEnd(16, "0");
1203
+ }
1204
+ function padTraceId$1(id) {
1205
+ return id.replace(/-/g, "").slice(0, 32).padEnd(32, "0");
1206
+ }
1207
+ //#endregion
1208
+ //#region src/trace/store-to-otlp.ts
1209
+ /**
1210
+ * Convert agent-eval's internal trace shape (`FileSystemTraceStore` → `Run`,
1211
+ * `Span`, `TraceEvent`) into the OTLP-flat JSONL the trace analyst
1212
+ * (`analyzeTraces` + `OtlpFileTraceStore`) reads.
1213
+ *
1214
+ * Eval harnesses shard a `FileSystemTraceStore` per cell (persona / variant)
1215
+ * under a run directory. The analyst consumes a single OTLP-NDJSON file keyed
1216
+ * on `trace_id` + `span_id` with `start_time`/`end_time` in ISO-8601 and
1217
+ * resource + `attributes` rolled up per-span. This module walks every shard,
1218
+ * projects each `Span` (plus events pinned to it) into the flat OTLP shape,
1219
+ * and emits one NDJSON file.
1220
+ *
1221
+ * Generic OTLP/OpenInference fields are always emitted (`service.name`,
1222
+ * `agent.name`, `run.id`/`run.status`, `openinference.span.kind`,
1223
+ * `llm.model_name`, …). Domain attributes (`legal.*`, `tax.*`, …) are injected
1224
+ * per-run via {@link TraceStoreToOtlpOptions.resourceAttributes} /
1225
+ * {@link TraceStoreToOtlpOptions.runAttributes} so consumers don't re-roll the
1226
+ * walker.
1227
+ */
1228
+ /**
1229
+ * Read every per-cell shard under each source root and write a flat OTLP-JSONL
1230
+ * view of the corpus to `outPath`. Each cell directory is a
1231
+ * `FileSystemTraceStore` — NDJSON append-only with size-based rotation;
1232
+ * `updateRun`/`updateSpan` append `{ id, ...patch, _update: true }` rows
1233
+ * rather than rewriting, so readers must merge those patches in (done here).
1234
+ *
1235
+ * A `string` source is treated as a celled root.
1236
+ */
1237
+ function convertTraceStoresToOtlp(source, outPath, opts = {}) {
1238
+ const sources = Array.isArray(source) ? [...source] : typeof source === "string" ? [{
1239
+ root: source,
1240
+ layout: "celled"
1241
+ }] : [source];
1242
+ const defaultServiceName = opts.serviceName ?? "agent-eval";
1243
+ const resourceAttributes = opts.resourceAttributes ?? (() => ({}));
1244
+ const runAttributes = opts.runAttributes ?? (() => ({}));
1245
+ const lines = [];
1246
+ let spanCount = 0;
1247
+ let runCount = 0;
1248
+ let cellCount = 0;
1249
+ let cellErrorCount = 0;
1250
+ for (const src of sources) {
1251
+ const serviceName = src.serviceName ?? defaultServiceName;
1252
+ const cellDirs = src.layout === "flat" ? [{
1253
+ label: "<root>",
1254
+ dir: src.root
1255
+ }] : listCells(src.root).map((name) => ({
1256
+ label: name,
1257
+ dir: join(src.root, name)
1258
+ }));
1259
+ for (const cell of cellDirs) try {
1260
+ const result = projectCell({
1261
+ cellDir: cell.dir,
1262
+ serviceName,
1263
+ resourceAttributes,
1264
+ runAttributes
1265
+ });
1266
+ for (const line of result.lines) lines.push(line);
1267
+ spanCount += result.spanCount;
1268
+ runCount += result.runCount;
1269
+ cellCount += 1;
1270
+ } catch (err) {
1271
+ console.warn(`[traces-to-otlp] cell ${cell.label} (${cell.dir}) skipped: ${err instanceof Error ? err.message : String(err)}`);
1272
+ cellErrorCount += 1;
1273
+ }
1274
+ }
1275
+ writeFileSync(outPath, lines.join("\n") + (lines.length > 0 ? "\n" : ""));
1276
+ return {
1277
+ spanCount,
1278
+ runCount,
1279
+ cellCount,
1280
+ cellErrorCount
1281
+ };
1282
+ }
1283
+ function projectCell(args) {
1284
+ const { cellDir, serviceName, resourceAttributes, runAttributes } = args;
1285
+ const lines = [];
1286
+ let runCount = 0;
1287
+ let spanCount = 0;
1288
+ const runs = readMergedShards(cellDir, "runs", "runId");
1289
+ const spans = readMergedShards(cellDir, "spans", "spanId");
1290
+ const events = readShards(cellDir, "events");
1291
+ const runByRunId = /* @__PURE__ */ new Map();
1292
+ for (const r of runs) runByRunId.set(r.runId, r);
1293
+ const spanBySpanId = /* @__PURE__ */ new Map();
1294
+ for (const s of spans) spanBySpanId.set(s.spanId, s);
1295
+ const eventsBySpanId = /* @__PURE__ */ new Map();
1296
+ for (const e of events) {
1297
+ if (e.kind === "state_mutation" && e.payload && typeof e.payload === "object") {
1298
+ const entity = e.payload.entity;
1299
+ if (entity === "run") {
1300
+ const run = e.payload.run;
1301
+ if (run?.runId) runByRunId.set(run.runId, run);
1302
+ continue;
1303
+ }
1304
+ if (entity === "run.update") {
1305
+ const patch = e.payload.patch;
1306
+ if (patch && e.runId) {
1307
+ const prior = runByRunId.get(e.runId);
1308
+ if (prior) runByRunId.set(e.runId, {
1309
+ ...prior,
1310
+ ...patch
1311
+ });
1312
+ }
1313
+ continue;
1314
+ }
1315
+ if (entity === "span") {
1316
+ const span = e.payload.span;
1317
+ if (span?.spanId) spanBySpanId.set(span.spanId, span);
1318
+ continue;
1319
+ }
1320
+ if (entity === "span.update") {
1321
+ const spanId = e.payload.spanId;
1322
+ const patch = e.payload.patch;
1323
+ if (spanId && patch) {
1324
+ const prior = spanBySpanId.get(spanId);
1325
+ if (prior) spanBySpanId.set(spanId, {
1326
+ ...prior,
1327
+ ...patch
1328
+ });
1329
+ }
1330
+ continue;
1331
+ }
1332
+ }
1333
+ if (!e.spanId) continue;
1334
+ const arr = eventsBySpanId.get(e.spanId) ?? [];
1335
+ arr.push(e);
1336
+ eventsBySpanId.set(e.spanId, arr);
1337
+ }
1338
+ for (const run of runByRunId.values()) {
1339
+ const traceId = padTraceId(run.runId);
1340
+ const agentName = run.variantId ?? run.scenarioId;
1341
+ const sharedResource = { attributes: {
1342
+ "service.name": serviceName,
1343
+ "agent.name": agentName,
1344
+ "run.id": run.runId,
1345
+ "run.status": run.status,
1346
+ ...resourceAttributes(run)
1347
+ } };
1348
+ const runSpanId = padSpanId(`run-${run.runId}`);
1349
+ const runStart = msToIso(run.startedAt);
1350
+ const runEnd = msToIso(run.endedAt ?? run.startedAt);
1351
+ const runStatus = run.outcome?.failureClass && run.outcome.failureClass !== "success" ? "STATUS_CODE_ERROR" : "STATUS_CODE_OK";
1352
+ const runAttrs = {
1353
+ [OPENINFERENCE_SPAN_KIND]: "AGENT",
1354
+ "agent.name": agentName,
1355
+ "agent.workflow.name": serviceName,
1356
+ ...runAttributes(run)
1357
+ };
1358
+ lines.push(JSON.stringify(toLine({
1359
+ traceId,
1360
+ spanId: runSpanId,
1361
+ parentSpanId: "",
1362
+ name: `run.${agentName}`,
1363
+ kind: "SPAN_KIND_INTERNAL",
1364
+ startTime: runStart,
1365
+ endTime: runEnd,
1366
+ statusCode: runStatus,
1367
+ statusMessage: run.outcome?.notes ?? "",
1368
+ resource: sharedResource,
1369
+ attributes: runAttrs
1370
+ })));
1371
+ runCount += 1;
1372
+ for (const span of spanBySpanId.values()) {
1373
+ if (span.runId !== run.runId) continue;
1374
+ const spanAttrs = spanToAttributes(span, eventsBySpanId.get(span.spanId) ?? []);
1375
+ const statusCode = span.status === "error" ? "STATUS_CODE_ERROR" : "STATUS_CODE_OK";
1376
+ lines.push(JSON.stringify(toLine({
1377
+ traceId,
1378
+ spanId: padSpanId(span.spanId),
1379
+ parentSpanId: span.parentSpanId ? padSpanId(span.parentSpanId) : runSpanId,
1380
+ name: span.name,
1381
+ kind: spanKindToOtlpKind(span.kind),
1382
+ startTime: msToIso(span.startedAt),
1383
+ endTime: msToIso(span.endedAt ?? span.startedAt),
1384
+ statusCode,
1385
+ statusMessage: span.error ?? "",
1386
+ resource: sharedResource,
1387
+ attributes: spanAttrs
1388
+ })));
1389
+ spanCount += 1;
1390
+ }
1391
+ }
1392
+ return {
1393
+ lines,
1394
+ runCount,
1395
+ spanCount
1396
+ };
1397
+ }
1398
+ function listCells(root) {
1399
+ try {
1400
+ return readdirSync(root, { withFileTypes: true }).filter((d) => d.isDirectory()).map((d) => d.name).sort();
1401
+ } catch {
1402
+ return [];
1403
+ }
1404
+ }
1405
+ /**
1406
+ * Read every NDJSON shard for `name` under `cellDir`, ordered by mtime so
1407
+ * rotated files apply before the active one. Yields raw rows including any
1408
+ * `_update: true` patches.
1409
+ */
1410
+ function readShards(cellDir, name) {
1411
+ let entries;
1412
+ try {
1413
+ entries = readdirSync(cellDir);
1414
+ } catch {
1415
+ return [];
1416
+ }
1417
+ const shards = entries.filter((f) => (f === `${name}.ndjson` || f.startsWith(`${name}.`)) && f.endsWith(".ndjson")).map((f) => ({
1418
+ file: f,
1419
+ path: join(cellDir, f)
1420
+ })).map((s) => {
1421
+ let mtime = 0;
1422
+ try {
1423
+ mtime = statSync(s.path).mtimeMs;
1424
+ } catch {}
1425
+ return {
1426
+ ...s,
1427
+ mtime
1428
+ };
1429
+ }).sort((a, b) => a.mtime - b.mtime || a.file.localeCompare(b.file));
1430
+ const rows = [];
1431
+ for (const shard of shards) {
1432
+ let text;
1433
+ try {
1434
+ text = readFileSync(shard.path, "utf-8");
1435
+ } catch {
1436
+ continue;
1437
+ }
1438
+ for (const line of text.split("\n")) {
1439
+ const trimmed = line.trim();
1440
+ if (!trimmed) continue;
1441
+ try {
1442
+ rows.push(JSON.parse(trimmed));
1443
+ } catch {}
1444
+ }
1445
+ }
1446
+ return rows;
1447
+ }
1448
+ /**
1449
+ * Read NDJSON shards and merge `{ ...patch, _update: true }` rows into the
1450
+ * prior record keyed on `idKey` — mirrors the in-memory merge
1451
+ * `FileSystemTraceStore` keeps but doesn't replay on cross-process load.
1452
+ */
1453
+ function readMergedShards(cellDir, name, idKey) {
1454
+ const rows = readShards(cellDir, name);
1455
+ const byId = /* @__PURE__ */ new Map();
1456
+ for (const row of rows) {
1457
+ const id = row[idKey];
1458
+ if (!id) continue;
1459
+ const prior = byId.get(id);
1460
+ if (prior && row._update) byId.set(id, {
1461
+ ...prior,
1462
+ ...row,
1463
+ _update: void 0
1464
+ });
1465
+ else byId.set(id, row);
1466
+ }
1467
+ return [...byId.values()];
1468
+ }
1469
+ function spanToAttributes(span, events) {
1470
+ const attrs = { [OPENINFERENCE_SPAN_KIND]: traceSpanKindToOpenInferenceKind(span.kind) };
1471
+ if (span.kind === "llm") {
1472
+ applyLlmSpanOtlpAttributes(attrs, span);
1473
+ if (Array.isArray(span.messages)) attrs["llm.input_messages"] = JSON.stringify(span.messages.slice(-6));
1474
+ if (typeof span.output === "string") attrs["llm.output_messages"] = JSON.stringify([{
1475
+ role: "assistant",
1476
+ content: span.output
1477
+ }]);
1478
+ } else if (span.kind === "tool") applyToolSpanOtlpAttributes(attrs, span);
1479
+ else if (span.kind === "judge") {
1480
+ attrs["judge.id"] = span.judgeId;
1481
+ attrs["judge.dimension"] = span.dimension;
1482
+ attrs["judge.score"] = span.score;
1483
+ attrs["judge.target_span_id"] = span.targetSpanId;
1484
+ }
1485
+ if (span.attributes) for (const [k, v] of Object.entries(span.attributes)) attrs[`agent_eval.${k}`] = v;
1486
+ if (events.length > 0) {
1487
+ attrs["agent_eval.event_count"] = events.length;
1488
+ attrs["agent_eval.event_kinds"] = JSON.stringify(events.map((e) => e.kind));
1489
+ }
1490
+ return attrs;
1491
+ }
1492
+ function spanKindToOtlpKind(kind) {
1493
+ switch (kind) {
1494
+ case "llm": return "SPAN_KIND_CLIENT";
1495
+ case "retrieval": return "SPAN_KIND_CLIENT";
1496
+ default: return "SPAN_KIND_INTERNAL";
1497
+ }
1498
+ }
1499
+ function toLine(args) {
1500
+ return {
1501
+ trace_id: args.traceId,
1502
+ span_id: args.spanId,
1503
+ parent_span_id: args.parentSpanId,
1504
+ name: args.name,
1505
+ kind: args.kind,
1506
+ start_time: args.startTime,
1507
+ end_time: args.endTime,
1508
+ status: {
1509
+ code: args.statusCode,
1510
+ message: args.statusMessage
1511
+ },
1512
+ resource: args.resource,
1513
+ attributes: args.attributes
1514
+ };
1515
+ }
1516
+ function msToIso(ms) {
1517
+ if (!Number.isFinite(ms) || ms <= 0) return (/* @__PURE__ */ new Date(0)).toISOString();
1518
+ return new Date(ms).toISOString();
1519
+ }
1520
+ /** OTLP wants 16-hex span ids; eval traces use UUID-ish strings. Hex-strip +
1521
+ * take 16, else deterministic FNV fold. */
1522
+ function padSpanId(id) {
1523
+ const cleaned = id.replace(/[^a-f0-9]/gi, "").toLowerCase();
1524
+ if (cleaned.length >= 16) return cleaned.slice(0, 16);
1525
+ return foldTo16Hex(id);
1526
+ }
1527
+ /** OTLP wants 32-hex trace ids; fold deterministically when too short. */
1528
+ function padTraceId(id) {
1529
+ const cleaned = id.replace(/[^a-f0-9]/gi, "").toLowerCase();
1530
+ if (cleaned.length >= 32) return cleaned.slice(0, 32);
1531
+ return foldTo32Hex(id);
1532
+ }
1533
+ function foldTo16Hex(s) {
1534
+ let h1 = 2166136261;
1535
+ for (const ch of s) {
1536
+ h1 ^= ch.charCodeAt(0);
1537
+ h1 = Math.imul(h1, 16777619) >>> 0;
1538
+ }
1539
+ const part = h1.toString(16).padStart(8, "0");
1540
+ return (part + part).slice(0, 16);
1541
+ }
1542
+ function foldTo32Hex(s) {
1543
+ return foldTo16Hex(s) + foldTo16Hex(`${s}::trace`).slice(0, 16);
1544
+ }
1545
+ //#endregion
1546
+ //#region src/replay.ts
1547
+ /**
1548
+ * Replay-from-raw-events — turn every captured campaign run into a
1549
+ * re-runnable artifact.
1550
+ *
1551
+ * `RawProviderSink` captures every provider HTTP envelope; `runEvalCampaign`
1552
+ * makes that capture the default. Together they make every past run a
1553
+ * complete fingerprint of what happened on the wire — enough to replay
1554
+ * the run without burning new LLM cost.
1555
+ *
1556
+ * Three use cases this primitive enables:
1557
+ *
1558
+ * 1. **Post-hoc judging** — apply a new judge / rubric / scoring callback
1559
+ * to last week's runs without re-calling any LLM. The cost of trying
1560
+ * a new rubric drops from "another full sweep" to a CPU-bound replay.
1561
+ * 2. **Determinism audits** — replay the same campaign and verify the
1562
+ * raw responses match byte-for-byte. Any drift is a non-determinism
1563
+ * bug (in the harness, the prompt builder, the sandbox, …).
1564
+ * 3. **Free judge calibration** — run two judges on identical responses
1565
+ * and measure inter-judge agreement without doubling LLM spend.
1566
+ *
1567
+ * The interface is deliberately fetch-shaped. Inject `createReplayFetch`
1568
+ * into `LlmClientOptions.fetch` and every `callLlm` transparently reads
1569
+ * from the cache instead of calling the network. No new code path through
1570
+ * the LLM client is needed; the cache hit is invisible to the runner.
1571
+ */
1572
+ var ReplayCacheMissError = class extends ReplayError {
1573
+ url;
1574
+ requestKey;
1575
+ constructor(url, requestKey, message) {
1576
+ super(message ?? `replay cache miss for ${url} (key=${requestKey})`);
1577
+ this.url = url;
1578
+ this.requestKey = requestKey;
1579
+ }
1580
+ };
1581
+ /**
1582
+ * In-memory deterministic cache of (request → response) keyed on a stable
1583
+ * hash of the request body. Built from a `RawProviderSink` containing
1584
+ * paired `request` and `response` events from a previous run.
1585
+ *
1586
+ * The cache is the source of truth for replay; `createReplayFetch` is a
1587
+ * thin wrapper that reads from it.
1588
+ */
1589
+ var ReplayCache = class ReplayCache {
1590
+ byKey = /* @__PURE__ */ new Map();
1591
+ orphans = 0;
1592
+ byProvider = {};
1593
+ byModel = {};
1594
+ /**
1595
+ * Build a cache from a sink's events. The sink must implement `list()`.
1596
+ * Filter by `runId` / `spanId` to scope to a specific replay.
1597
+ */
1598
+ static async fromSink(sink, filter = {}) {
1599
+ if (!sink.list) throw new ReplayError("ReplayCache.fromSink: sink must implement list() to be replayable.");
1600
+ const events = await sink.list(filter);
1601
+ return ReplayCache.fromEvents(events);
1602
+ }
1603
+ /** Build a cache from an in-memory event list. */
1604
+ static async fromEvents(events) {
1605
+ const cache = new ReplayCache();
1606
+ const groups = /* @__PURE__ */ new Map();
1607
+ for (const e of events) {
1608
+ const k = `${e.runId ?? ""}::${e.spanId ?? ""}::${e.attemptIndex}`;
1609
+ const g = groups.get(k) ?? {};
1610
+ if (e.direction === "request") g.req = e;
1611
+ else g.res = e;
1612
+ groups.set(k, g);
1613
+ }
1614
+ for (const g of groups.values()) {
1615
+ if (!g.req) continue;
1616
+ if (!g.res) {
1617
+ cache.orphans += 1;
1618
+ continue;
1619
+ }
1620
+ const key = await requestKey(g.req);
1621
+ cache.byKey.set(key, {
1622
+ request: g.req,
1623
+ response: g.res
1624
+ });
1625
+ cache.byProvider[g.req.provider] = (cache.byProvider[g.req.provider] ?? 0) + 1;
1626
+ cache.byModel[g.req.model] = (cache.byModel[g.req.model] ?? 0) + 1;
1627
+ }
1628
+ return cache;
1629
+ }
1630
+ /** Number of cacheable (request, response) pairs in the cache. */
1631
+ size() {
1632
+ return this.byKey.size;
1633
+ }
1634
+ stats() {
1635
+ return {
1636
+ total: this.byKey.size,
1637
+ byProvider: { ...this.byProvider },
1638
+ byModel: { ...this.byModel },
1639
+ orphanRequests: this.orphans
1640
+ };
1641
+ }
1642
+ /** Iterate every cached `(request, response)` pair in insertion order. */
1643
+ *entries() {
1644
+ for (const entry of this.byKey.values()) yield entry;
1645
+ }
1646
+ /**
1647
+ * Look up a cached response by hashing the (model, messages, temperature,
1648
+ * maxTokens, response_format) shape. Returns `undefined` on miss; the
1649
+ * caller decides whether to throw, fall back to the network, or skip.
1650
+ */
1651
+ async lookup(requestBody) {
1652
+ const key = await keyFromBody(requestBody);
1653
+ return this.byKey.get(key);
1654
+ }
1655
+ };
1656
+ /**
1657
+ * Build a `fetch`-shaped function that serves cached responses out of a
1658
+ * `ReplayCache` for any URL ending in `/chat/completions`. Pass through
1659
+ * `LlmClientOptions.fetch` and `callLlm` becomes free.
1660
+ *
1661
+ * Non-`/chat/completions` URLs are passed straight to the fallback fetch
1662
+ * (default: `globalThis.fetch`). This matters because non-LLM HTTP work
1663
+ * (judge HTTP servers, sandbox callbacks) sometimes flows through the same
1664
+ * `fetch` and shouldn't be intercepted.
1665
+ */
1666
+ function createReplayFetch(cache, opts = {}) {
1667
+ const onMiss = opts.onMiss ?? "throw";
1668
+ const fallback = opts.fallbackFetch ?? globalThis.fetch?.bind(globalThis);
1669
+ return (async (input, init) => {
1670
+ const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
1671
+ if (!/\/chat\/completions(?:[?#].*)?$/.test(url)) {
1672
+ if (!fallback) throw new ReplayError(`replay fetch: non-completions URL ${url} but no fallbackFetch configured`);
1673
+ return fallback(input, init);
1674
+ }
1675
+ let bodyParsed;
1676
+ if (init?.body && typeof init.body === "string") try {
1677
+ bodyParsed = JSON.parse(init.body);
1678
+ } catch {}
1679
+ const hit = bodyParsed === void 0 ? void 0 : await cache.lookup(bodyParsed);
1680
+ if (hit) {
1681
+ opts.onHit?.({
1682
+ url,
1683
+ provider: hit.request.provider,
1684
+ model: hit.request.model
1685
+ });
1686
+ const status = hit.response.statusCode ?? 200;
1687
+ const headers = new Headers(Object.entries(hit.response.responseHeaders ?? { "Content-Type": "application/json" }));
1688
+ const bodyText = typeof hit.response.responseBody === "string" ? hit.response.responseBody : JSON.stringify(hit.response.responseBody ?? {});
1689
+ return new Response(bodyText, {
1690
+ status,
1691
+ headers
1692
+ });
1693
+ }
1694
+ opts.onMissNotify?.({
1695
+ url,
1696
+ requestBody: bodyParsed
1697
+ });
1698
+ if (onMiss === "throw") throw new ReplayCacheMissError(url, bodyParsed === void 0 ? "<unparseable>" : await keyFromBody(bodyParsed));
1699
+ if (onMiss === "fail-closed") return new Response(JSON.stringify({ error: "replay_cache_miss" }), { status: 599 });
1700
+ if (!fallback) throw new ReplayError("replay fetch: onMiss=fallback but no fallbackFetch configured");
1701
+ return fallback(input, init);
1702
+ });
1703
+ }
1704
+ /**
1705
+ * Convenience iterator over `(request, response)` pairs in a sink — for
1706
+ * post-hoc scoring that doesn't need a `fetch` shim. The judge or scorer
1707
+ * runs purely in-process over cached LLM outputs.
1708
+ */
1709
+ async function* iterateRawCalls(sink, filter = {}) {
1710
+ if (!sink.list) throw new ReplayError("iterateRawCalls: sink must implement list().");
1711
+ const events = await sink.list(filter);
1712
+ const cache = await ReplayCache.fromEvents(events);
1713
+ for (const entry of cache.entries()) yield entry;
1714
+ }
1715
+ /**
1716
+ * Canonical request key.
1717
+ *
1718
+ * `model + messages + temperature + max_tokens|max_completion_tokens +
1719
+ * response_format` are the dimensions that affect the response shape.
1720
+ * Other fields (timestamp headers, provider-specific metadata) are
1721
+ * intentionally excluded so a request hashes the same across re-runs.
1722
+ */
1723
+ async function requestKey(event) {
1724
+ return keyFromBody(event.requestBody);
1725
+ }
1726
+ async function keyFromBody(body) {
1727
+ if (body == null || typeof body !== "object") return hashJson({ raw: String(body) });
1728
+ const b = body;
1729
+ return hashJson(canonicalize({
1730
+ model: b.model ?? null,
1731
+ messages: b.messages ?? null,
1732
+ temperature: b.temperature ?? null,
1733
+ max_tokens: b.max_tokens ?? null,
1734
+ max_completion_tokens: b.max_completion_tokens ?? null,
1735
+ response_format: b.response_format ?? null
1736
+ }));
1737
+ }
1738
+ //#endregion
1739
+ export { planTraceInsightQuestions as C, traceAnalystOnRunComplete as E, inferDomainKeywords as S, tokenizeDomainWords as T, buildTraceInsightContext as _, convertTraceStoresToOtlp as a, describeTraceInsightScope as b, otelRunCompleteHook as c, captureFetchToRawSink as d, otlpRowsToRunRecords as f, flattenOtlpExportToNdjson as g, otlpToTraceRunRecords as h, iterateRawCalls as i, OTEL_AGENT_EVAL_SCOPE as l, otlpToRunRecords as m, ReplayCacheMissError as n, createOtelExporter as o, otlpRowsToTraceRunRecords as p, createReplayFetch as r, createOtelTracingStore as s, ReplayCache as t, exportRunAsOtlp as u, buildTraceInsightPrompt as v, scoreTraceInsightReadiness as w, domainEvidencePattern as x, defaultTraceInsightPanel as y };
1740
+
1741
+ //# sourceMappingURL=replay-GnyotH0J.js.map