@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -1,1859 +0,0 @@
1
- import { n as CaptureIntegrityError, s as ReplayError } from "./errors-D-LKuDhb.js";
2
- import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
3
- import { a as providerFromBaseUrl, i as defaultProviderRedactor } from "./raw-provider-sink-BQd7mzyT.js";
4
- import { LLM_INPUT_TOKENS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, OPENINFERENCE_SPAN_KIND, TOOL_NAME, applyLlmSpanOtlpAttributes } from "./trace-attributes.js";
5
- import { J as firstStringAttr, K as compareSpanTime, Q as spanEpochMillis, X as projectOtlpFlatLine, et as applyToolSpanOtlpAttributes, i as runTraceAnalyst, nt as isOtlpModelCall, rt as traceSpanKindToOpenInferenceKind, tt as classifyOtlpSpanRole } from "./kind-factory-B8-r8-y8.js";
6
- import { r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BmSPWXJR.js";
7
- import { n as OtlpFileTraceStore, r as createOtlpBufferTraceStore } from "./store-otlp-Dw8PPIlL.js";
8
- import { a as recordAggregateMeasurements, i as readTaskFailureLabels, o as summarizeExecutionMeasurements, r as extractUsageFromSse, s as summarizeTraceErrors, t as extractUsage } from "./extract-usage-CdZdoj1s.js";
9
- import { readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
10
- import { join } from "node:path";
11
- import { randomUUID } from "node:crypto";
12
- import { deriveHexId, isW3CSpanId, isW3CTraceId } from "@tangle-network/agent-trace-contract";
13
- //#region src/trace-analyst/prompts.ts
14
- /** General policy for recursive, evidence-backed trace analysis. */
15
- const TRACE_ANALYST_ACTOR_DESCRIPTION = `Answer the question by inspecting the OTLP trace dataset with the available tools.
16
-
17
- 1. Call getDatasetOverview first. Use its real trace ids and dataset size to plan the investigation.
18
- 2. Narrow with queryTraces and countTraces before scanning large payloads.
19
- 3. For a small trace, use viewTrace. For a large trace, use searchTrace and then viewSpans or searchSpan.
20
- 4. Never invent a trace id, span id, tool result, error, frequency, or final outcome.
21
- 5. When a search reports has_more, refine the query before drawing a conclusion.
22
- 6. Use llm_query only over evidence already loaded. A recursive query cannot inspect traces itself.
23
- 7. Cite exact evidence URIs returned by the tools. Include a short exact excerpt when it supports the claim.
24
- 8. Return no finding when the available evidence cannot support one.
25
-
26
- The prose answer must directly answer the question and state important uncertainty.
27
- The findings array contains only actionable or decision-relevant claims supported by inspected evidence.`;
28
- const TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION = "trace-analyst-research-v1-2026-07-30";
29
- //#endregion
30
- //#region src/trace-analyst/analyst.ts
31
- /**
32
- * Answer one question by recursively inspecting a trace store.
33
- *
34
- * The returned answer, cited findings, engine steps, call counts, and runtime
35
- * identity are one audit record. A direct one-shot model call is not used.
36
- */
37
- async function analyzeTraces(input, options) {
38
- if (typeof input.question !== "string" || !input.question.trim()) throw new TypeError("analyzeTraces: input.question must be a non-empty string");
39
- const id = input.id?.trim() || "trace-analysis";
40
- const store = typeof options.source === "string" ? new OtlpFileTraceStore({ path: options.source }) : options.source;
41
- if (store instanceof OtlpFileTraceStore) await store.ensureIndexed(options.signal ? { signal: options.signal } : void 0);
42
- return runTraceAnalyst({
43
- definition: {
44
- id,
45
- description: input.description?.trim() || "Answers a caller-defined question by recursively inspecting trace evidence.",
46
- area: input.area?.trim() || "trace-analysis",
47
- version: TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
48
- question: input.question,
49
- instructions: options.instructions ?? TRACE_ANALYST_ACTOR_DESCRIPTION,
50
- toolGroup: options.toolGroup ?? "all",
51
- limits: options.limits
52
- },
53
- engine: options.engine,
54
- store,
55
- context: {
56
- runId: options.runId ?? id,
57
- correlationId: randomUUID(),
58
- budgetUsd: options.budgetUsd,
59
- costLedger: options.costLedger,
60
- costPhase: options.costPhase ?? "trace-analysis",
61
- priorFindings: options.priorFindings,
62
- upstreamFindings: options.upstreamFindings,
63
- recordUsage: options.recordUsage,
64
- tags: options.tags,
65
- log: options.log,
66
- signal: options.signal
67
- }
68
- });
69
- }
70
- //#endregion
71
- //#region src/trace-analyst/hook.ts
72
- const DEFAULT_QUESTION = "Summarise what happened in this run. Surface any failure modes, surprising findings, or evidence that the run's verdict is wrong.";
73
- function traceAnalystOnRunComplete(opts) {
74
- return async (ctx) => {
75
- if (opts.shouldRun && !opts.shouldRun(ctx)) return;
76
- const source = opts.analyze.source;
77
- if (source === void 0) {
78
- await ctx.store.appendEvent({
79
- eventId: `analyst-skip-${ctx.runId}`,
80
- runId: ctx.runId,
81
- kind: "log",
82
- timestamp: Date.now(),
83
- payload: {
84
- source: "trace_analyst_hook",
85
- reason: "no source configured"
86
- }
87
- });
88
- return;
89
- }
90
- const result = await analyzeTraces({ question: opts.question ?? DEFAULT_QUESTION }, {
91
- ...opts.analyze,
92
- source
93
- });
94
- if (opts.save) await opts.save(result, ctx);
95
- if (opts.gateOn && !opts.gateOn(result, ctx)) await ctx.store.appendEvent({
96
- eventId: `analyst-gate-${ctx.runId}`,
97
- runId: ctx.runId,
98
- kind: "log",
99
- timestamp: Date.now(),
100
- payload: {
101
- source: "trace_analyst_hook",
102
- reason: "analyst_gate_failed",
103
- findings: result.findings
104
- }
105
- });
106
- };
107
- }
108
- //#endregion
109
- //#region src/trace-analyst/insights.ts
110
- const DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
111
- "and",
112
- "advanced",
113
- "app",
114
- "build",
115
- "create",
116
- "easy",
117
- "expert",
118
- "extreme",
119
- "for",
120
- "from",
121
- "hard",
122
- "implementation",
123
- "integrate",
124
- "medium",
125
- "project",
126
- "task",
127
- "the",
128
- "this",
129
- "with",
130
- "workflow"
131
- ]);
132
- function tokenizeDomainWords(value) {
133
- return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
134
- }
135
- function inferDomainKeywords(suite) {
136
- const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
137
- const source = [
138
- suite.name,
139
- suite.collectionId ?? "",
140
- ...suite.tasks.flatMap((task) => [
141
- task.id,
142
- task.name,
143
- task.prompt ?? "",
144
- task.difficulty ?? "",
145
- ...task.tags ?? [],
146
- ...task.gaps ?? []
147
- ])
148
- ].join(" ");
149
- const counts = /* @__PURE__ */ new Map();
150
- for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
151
- return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
152
- }
153
- function domainEvidencePattern(keywords) {
154
- const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
155
- return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
156
- }
157
- function describeTraceInsightScope(suite) {
158
- const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
159
- const tags = /* @__PURE__ */ new Map();
160
- for (const task of suite.tasks) for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
161
- const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
162
- if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
163
- return `${taskLabel} across ${[...new Set(suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value)))].join(", ") || "the selected benchmark scope"}.`;
164
- }
165
- function planTraceInsightQuestions(input) {
166
- const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
167
- const hasMultipleShots = input.suite.tasks.some((task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap)));
168
- const questions = [
169
- {
170
- id: "execution-path",
171
- question: "What did the worker actually do before the first meaningful implementation edit?",
172
- why: "Separates grounded execution from polished but shallow output."
173
- },
174
- {
175
- id: "research-grounding",
176
- question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
177
- why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
178
- },
179
- {
180
- id: "domain-proof",
181
- question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
182
- why: "Keeps product-quality claims tied to concrete evidence."
183
- },
184
- {
185
- id: "root-cause",
186
- question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
187
- why: "Turns trace observations into actionable ownership."
188
- },
189
- {
190
- id: "evidence-quality",
191
- question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
192
- why: "Prevents unsupported customer-report conclusions."
193
- }
194
- ];
195
- if (hasMultipleShots) questions.push({
196
- id: "reviewer-lift",
197
- question: "Where did reviewer feedback improve score, stall, or regress across shots?",
198
- why: "Shows whether the driver loop is learning or merely repeating work."
199
- });
200
- if (hasFailures) questions.push({
201
- id: "optimization-targets",
202
- question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
203
- why: "Connects benchmark evidence to the optimization loop."
204
- });
205
- return questions;
206
- }
207
- function buildTraceInsightContext(input) {
208
- return {
209
- suite: input.suite,
210
- scope: describeTraceInsightScope(input.suite),
211
- keywords: inferDomainKeywords(input.suite),
212
- questions: planTraceInsightQuestions(input),
213
- panel: defaultTraceInsightPanel(),
214
- findings: input.findings ?? [],
215
- agent: input.agent ?? null,
216
- totals: input.totals ?? null
217
- };
218
- }
219
- function scoreTraceInsightReadiness(context) {
220
- const failedTasks = context.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied");
221
- const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
222
- const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
223
- const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
224
- const gates = [
225
- {
226
- id: "domain-context",
227
- label: "Domain context inferred",
228
- passed: context.keywords.length > 0,
229
- severity: "high",
230
- detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
231
- },
232
- {
233
- id: "panel-coverage",
234
- label: "Analyst panel planned",
235
- passed: context.panel.length >= 4 && context.questions.length >= 5,
236
- severity: "high",
237
- detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
238
- },
239
- {
240
- id: "failure-coverage",
241
- label: "Failures mapped to findings",
242
- passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= .5,
243
- severity: "critical",
244
- detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
245
- },
246
- {
247
- id: "gap-evidence",
248
- label: "Task gaps captured",
249
- passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= .5,
250
- severity: "medium",
251
- detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
252
- }
253
- ];
254
- const penalty = gates.reduce((sum, gate) => {
255
- if (gate.passed) return sum;
256
- if (gate.severity === "critical") return sum + 35;
257
- if (gate.severity === "high") return sum + 20;
258
- if (gate.severity === "medium") return sum + 10;
259
- return sum + 5;
260
- }, 0);
261
- const score = Math.max(0, Math.min(1, 1 - penalty / 100));
262
- return {
263
- score,
264
- grade: score >= .9 ? "external-ready" : score >= .7 ? "internal-review" : "raw-analysis",
265
- gates
266
- };
267
- }
268
- function defaultTraceInsightPanel() {
269
- return [
270
- {
271
- id: "trace-forensics",
272
- name: "Trace Forensics",
273
- responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
274
- },
275
- {
276
- id: "root-cause",
277
- name: "Root Cause",
278
- responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
279
- },
280
- {
281
- id: "optimization",
282
- name: "Optimization",
283
- responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
284
- },
285
- {
286
- id: "external-evidence",
287
- name: "External Evidence",
288
- responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
289
- }
290
- ];
291
- }
292
- function buildTraceInsightPrompt(input) {
293
- const context = buildTraceInsightContext(input);
294
- const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
295
- return `Analyze this benchmark run and produce evidence-backed trace intelligence.
296
-
297
- Audience:
298
- - internal AI/product leadership
299
- - possible customer-facing report for ${input.suite.name}
300
-
301
- Investigation plan:
302
- ${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
303
-
304
- Analyst panel:
305
- ${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
306
-
307
- If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
308
-
309
- Required output:
310
- 1. Executive verdict: what this run proves and does not prove.
311
- 2. The investigation questions you answered and the evidence used.
312
- 3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
313
- 4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
314
- 5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
315
- 6. What is safe for an external report versus what must stay internal.
316
- 7. One rerun plan that would validate lift after optimization.
317
-
318
- Budget:
319
- - Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
320
- - Prefer traces named in the failure summary over broad exploration.
321
- - Do not do exhaustive trace sweeps.
322
- - Return the final report as soon as the taxonomy and examples are supported.
323
-
324
- Run summary:
325
- ${JSON.stringify({
326
- suite: input.suite.name,
327
- scope: context.scope,
328
- inferredKeywords: context.keywords,
329
- agent: context.agent,
330
- totals: context.totals,
331
- findings: context.findings.map((finding) => ({
332
- kind: finding.kind,
333
- severity: finding.severity,
334
- taskCount: finding.taskIds.length,
335
- proposedFixClass: finding.proposedFixClass
336
- })),
337
- failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
338
- task: task.id,
339
- difficulty: task.difficulty,
340
- outcome: task.outcome,
341
- score: task.score,
342
- gaps: task.gaps ?? []
343
- }))
344
- }, null, 2)}
345
-
346
- Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
347
- }
348
- //#endregion
349
- //#region src/trace/otlp-flat.ts
350
- function createOtlpFlatLine(input) {
351
- return {
352
- trace_id: input.traceId,
353
- span_id: input.spanId,
354
- parent_span_id: input.parentSpanId,
355
- name: input.name,
356
- kind: input.kind,
357
- start_time: input.startTime,
358
- end_time: input.endTime,
359
- status: {
360
- code: input.statusCode,
361
- ...input.statusMessage !== void 0 ? { message: input.statusMessage } : {}
362
- },
363
- resource: input.resource,
364
- attributes: input.attributes,
365
- ...input.events && input.events.length > 0 ? { events: input.events } : {}
366
- };
367
- }
368
- /** Map the canonical trace status while letting each caller choose its legacy default. */
369
- function spanStatusToOtlp(status, error, defaultCode) {
370
- if (status === "error" || error) return "STATUS_CODE_ERROR";
371
- if (status === "ok") return "STATUS_CODE_OK";
372
- return defaultCode;
373
- }
374
- /** Convert epoch milliseconds, returning `undefined` for invalid or out-of-range values. */
375
- function epochMillisToIso(value) {
376
- if (!Number.isFinite(value)) return void 0;
377
- try {
378
- return new Date(value).toISOString();
379
- } catch {
380
- return;
381
- }
382
- }
383
- //#endregion
384
- //#region src/trace-analyst/otlp-flatten.ts
385
- const DEFAULT_KIND_MAP = {
386
- 0: "SPAN_KIND_UNSPECIFIED",
387
- 1: "SPAN_KIND_INTERNAL",
388
- 2: "SPAN_KIND_SERVER",
389
- 3: "SPAN_KIND_CLIENT",
390
- 4: "SPAN_KIND_PRODUCER",
391
- 5: "SPAN_KIND_CONSUMER"
392
- };
393
- const STATUS_MAP = {
394
- 0: "STATUS_CODE_UNSET",
395
- 1: "STATUS_CODE_OK",
396
- 2: "STATUS_CODE_ERROR"
397
- };
398
- /** Unwrap an OTLP attribute-value union to a scalar. */
399
- function attrValue(v) {
400
- if (v.stringValue !== void 0) return v.stringValue;
401
- if (v.intValue !== void 0) return Number(v.intValue);
402
- if (v.doubleValue !== void 0) return v.doubleValue;
403
- if (v.boolValue !== void 0) return v.boolValue;
404
- return "";
405
- }
406
- function attrsToRecord(attrs) {
407
- const out = {};
408
- for (const a of attrs) out[a.key] = attrValue(a.value);
409
- return out;
410
- }
411
- function nanoToIso(nano) {
412
- const ms = Number(nano) / 1e6;
413
- return Number.isFinite(ms) ? new Date(ms).toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
414
- }
415
- /** Mirror selected attributes into the OpenInference vocabulary in place. */
416
- function applyOpenInference(attrs) {
417
- if ("llm.model" in attrs && !("llm.model_name" in attrs)) attrs[LLM_MODEL_NAME] = attrs["llm.model"];
418
- if ("llm.input_tokens" in attrs && !("llm.token_count.prompt" in attrs)) attrs[LLM_INPUT_TOKENS] = attrs["llm.input_tokens"];
419
- if ("inference.llm.input_tokens" in attrs && !("llm.token_count.prompt" in attrs)) attrs[LLM_INPUT_TOKENS] = attrs["inference.llm.input_tokens"];
420
- if ("llm.output_tokens" in attrs && !("llm.token_count.completion" in attrs)) attrs[LLM_OUTPUT_TOKENS] = attrs["llm.output_tokens"];
421
- if ("inference.llm.output_tokens" in attrs && !("llm.token_count.completion" in attrs)) attrs[LLM_OUTPUT_TOKENS] = attrs["inference.llm.output_tokens"];
422
- if ("tool.name" in attrs && !("inference.tool.name" in attrs)) attrs["inference.tool.name"] = attrs[TOOL_NAME];
423
- if ("span.kind" in attrs && !("openinference.span.kind" in attrs)) attrs[OPENINFERENCE_SPAN_KIND] = String(attrs["span.kind"]).toUpperCase();
424
- }
425
- function flattenOtlpExportToNdjson(otlpExport, opts = {}) {
426
- const vocab = opts.attributeVocabulary ?? "openinference";
427
- const kindMap = {
428
- ...DEFAULT_KIND_MAP,
429
- ...opts.kindMap
430
- };
431
- const lines = [];
432
- for (const rs of otlpExport.resourceSpans ?? []) {
433
- const resource = { attributes: attrsToRecord(rs.resource?.attributes ?? []) };
434
- for (const scope of rs.scopeSpans ?? []) for (const span of scope.spans ?? []) {
435
- const attributes = attrsToRecord(span.attributes ?? []);
436
- if (vocab === "openinference") applyOpenInference(attributes);
437
- const line = createOtlpFlatLine({
438
- traceId: span.traceId,
439
- spanId: span.spanId,
440
- parentSpanId: span.parentSpanId ?? null,
441
- name: span.name,
442
- kind: kindMap[span.kind] ?? "SPAN_KIND_UNSPECIFIED",
443
- startTime: nanoToIso(span.startTimeUnixNano),
444
- endTime: nanoToIso(span.endTimeUnixNano),
445
- statusCode: STATUS_MAP[span.status?.code ?? 0] ?? "STATUS_CODE_UNSET",
446
- statusMessage: span.status?.message,
447
- resource,
448
- attributes,
449
- events: span.events?.map((e) => ({
450
- name: e.name,
451
- timeUnixNano: e.timeUnixNano,
452
- ...e.attributes ? { attributes: attrsToRecord(e.attributes) } : {}
453
- }))
454
- });
455
- lines.push(line);
456
- }
457
- }
458
- return lines;
459
- }
460
- //#endregion
461
- //#region src/trace-analyst/otlp-to-run-records.ts
462
- /**
463
- * `otlpToRunRecords` — fold an OTLP traces.jsonl (one OTLP span per line;
464
- * the form AppWorld / HALO emit via their OpenInference OTLP exporter, the
465
- * same shape `flattenOtlpExportToNdjson` produces) into validated
466
- * `RunRecord[]` — one record per `trace_id` by default, or per caller-defined
467
- * logical run when one task is fragmented across multiple OTLP traces.
468
- *
469
- * This is the offline ingestion primitive the AppWorld proposer bench and the
470
- * hosted Intelligence product both stand on: traces in, paper-grade rows
471
- * out, ready for `compareOptimizationMethods` / `analyzeRuns` / the promotion gate.
472
- *
473
- * Aggregation per trace:
474
- * - tokenUsage: reconcile input, output, cache-read, and cache-write across
475
- * nested model-call wrappers without double-counting parent aggregates.
476
- * - costUsd: reconcile complete observed model-call cost when present; else priced via
477
- * `opts.priceUsdPerToken` from the aggregated tokens; else `null` with a
478
- * loud `raw.cost_unpriced = 1` marker.
479
- * - task failure class and detail: read from process-root
480
- * `tangle.task.failure_*` attributes; malformed or conflicting values throw.
481
- * - terminalFailureReason: the failed root's normalized status message,
482
- * when one unambiguous root supplies terminal failure evidence.
483
- * - terminalOutcome: reduced from root-span status only. Child tool errors
484
- * remain visible in `error_span_count` and `execution_error_count` without
485
- * changing the run outcome. Root, guardrail, evaluator, propagated, and
486
- * unknown errors retain separate counters.
487
- * - model: the dominant LLM model in the trace (snapshot-padded to satisfy
488
- * `validateRunRecord` when the trace's model is a bare alias).
489
- * - outcome score: `opts.scoreForTrace` (AppWorld `world.evaluate()` →
490
- * TGC/SGC) when supplied. Traces without an external task-quality signal
491
- * remain unlabeled; execution errors never become a task score.
492
- * - prompt / completion: carried into `raw` as token-count signals and,
493
- * when the first/last LLM span exposes `input.value` / `output.value`,
494
- * the verbatim text is preserved on the optional `promptText` /
495
- * `completionText` of the returned `OtlpTraceRunRecord`.
496
- *
497
- * Fail-loud: an OTLP file with zero valid spans throws. A trace with no
498
- * spans is impossible (a trace exists only because a span referenced it).
499
- * `validateRunRecord` runs on every row — a malformed projection throws
500
- * rather than silently producing a half-record.
501
- */
502
- /**
503
- * Parse + aggregate an OTLP traces.jsonl string into validated
504
- * `RunRecord[]` (one per trace). Use {@link otlpToTraceRunRecords} when you
505
- * also want the verbatim prompt/completion text alongside each record.
506
- */
507
- function otlpToRunRecords(otlpJsonl, opts) {
508
- return otlpToTraceRunRecords(otlpJsonl, opts).map((r) => r.record);
509
- }
510
- /**
511
- * Aggregate already-parsed OTLP flat rows without serializing them back to
512
- * JSONL. This is the in-memory counterpart to {@link otlpToRunRecords}; both
513
- * paths share projection, reconciliation, validation, and ordering.
514
- */
515
- function otlpRowsToRunRecords(rows, opts) {
516
- return otlpRowsToTraceRunRecords(rows, opts).map((row) => row.record);
517
- }
518
- /** As {@link otlpToRunRecords} but returns the prompt/completion text too. */
519
- function otlpToTraceRunRecords(otlpJsonl, opts) {
520
- return traceRunRecordsFromSpans(groupSpansByLogicalRun(groupJsonlSpansByTrace(otlpJsonl), opts.logicalRunIdForTrace), opts);
521
- }
522
- /** Parsed-row counterpart to {@link otlpToTraceRunRecords}. */
523
- function otlpRowsToTraceRunRecords(rows, opts) {
524
- return traceRunRecordsFromSpans(groupSpansByLogicalRun(groupRowsByTrace(rows), opts.logicalRunIdForTrace), opts);
525
- }
526
- function traceRunRecordsFromSpans(byTrace, opts) {
527
- const splitTag = opts.splitTag ?? "holdout";
528
- const commitSha = opts.commitSha ?? "unknown";
529
- const promptHash = opts.promptHash ?? "unknown";
530
- const configHash = opts.configHash ?? "unknown";
531
- const seed = opts.seed ?? 0;
532
- const fallbackModel = opts.fallbackModel ?? "unknown@otlp";
533
- if (byTrace.size === 0) throw new Error("otlpToRunRecords: OTLP input produced zero valid spans — every row was empty, malformed, or missing trace_id/span_id");
534
- const traceIds = [...byTrace.keys()].sort();
535
- const out = [];
536
- for (const traceId of traceIds) {
537
- const spans = byTrace.get(traceId);
538
- const agg = aggregateTrace(traceId, spans, fallbackModel);
539
- const score = resolveScore(opts, traceId, agg);
540
- const { costUsd, costProvenance } = resolveCost(opts, agg);
541
- const raw = {
542
- source_trace_count: agg.sourceTraceCount,
543
- span_count: agg.spanCount,
544
- llm_span_count: agg.llmSpanCount,
545
- tool_span_count: agg.toolSpanCount,
546
- agent_span_count: agg.agentSpanCount,
547
- error_span_count: agg.errorSpanCount,
548
- execution_error_count: agg.executionErrorCount,
549
- process_error_count: agg.processErrorCount,
550
- guardrail_error_count: agg.guardrailErrorCount,
551
- judge_error_count: agg.judgeErrorCount,
552
- propagated_error_count: agg.propagatedErrorCount,
553
- unclassified_error_count: agg.unclassifiedErrorCount,
554
- prompt_tokens: agg.tokenUsage.input,
555
- completion_tokens: agg.tokenUsage.output
556
- };
557
- if (agg.tokenUsage.reasoning !== void 0) raw.reasoning_tokens = agg.tokenUsage.reasoning;
558
- if (agg.tokenUsage.cached !== void 0) raw.cached_tokens = agg.tokenUsage.cached;
559
- if (agg.tokenUsage.cacheWrite !== void 0) raw.cache_write_tokens = agg.tokenUsage.cacheWrite;
560
- if (agg.costMeasurement.value !== void 0 && !agg.costMeasurement.complete) raw.partial_observed_cost_usd = agg.costMeasurement.value;
561
- recordAggregateMeasurements(raw, agg.aggregateMeasurement);
562
- if (costProvenance.kind === "uncaptured") raw.cost_unpriced = 1;
563
- const outcome = { raw };
564
- if (score !== void 0) if (splitTag === "holdout") outcome.holdoutScore = score;
565
- else outcome.searchScore = score;
566
- const { promptText, completionText } = extractPromptCompletion(spans, agg.callSpanIds);
567
- const judgeMetadata = opts.judgeMetadataForTrace?.(traceId);
568
- const taskFailure = readTaskFailureLabels(spans.filter((span) => span.parent_span_id === null && isTerminalRootCandidate(span)), `otlpToRunRecords: run '${traceId}'`);
569
- const record = validateRunRecord({
570
- runId: `otlp:${opts.experimentId}:${opts.candidateId}:${traceId}`,
571
- experimentId: opts.experimentId,
572
- candidateId: opts.candidateId,
573
- seed,
574
- model: ensureSnapshot(agg.model, fallbackModel),
575
- promptHash,
576
- configHash,
577
- commitSha,
578
- wallMs: agg.wallMs,
579
- costUsd,
580
- costProvenance,
581
- tokenUsage: agg.tokenUsage,
582
- terminalOutcome: agg.terminalOutcome,
583
- ...agg.terminalFailureMessage ? { terminalFailureReason: agg.terminalFailureMessage } : {},
584
- ...judgeMetadata ? { judgeMetadata } : {},
585
- outcome,
586
- ...taskFailure,
587
- splitTag,
588
- scenarioId: traceId
589
- });
590
- out.push({
591
- record,
592
- ...promptText !== void 0 ? { promptText } : {},
593
- ...completionText !== void 0 ? { completionText } : {}
594
- });
595
- }
596
- return out;
597
- }
598
- function* yieldJsonlRows(otlpJsonl) {
599
- for (const line of otlpJsonl.split("\n")) {
600
- const trimmed = line.trim();
601
- if (trimmed.length === 0) continue;
602
- let parsed;
603
- try {
604
- parsed = JSON.parse(trimmed);
605
- } catch {
606
- continue;
607
- }
608
- if (parsed && typeof parsed === "object") yield parsed;
609
- }
610
- }
611
- function groupJsonlSpansByTrace(otlpJsonl) {
612
- return groupRowsByTrace(yieldJsonlRows(otlpJsonl));
613
- }
614
- function groupRowsByTrace(rows) {
615
- const byTrace = /* @__PURE__ */ new Map();
616
- for (const row of rows) {
617
- if (!row || typeof row !== "object") continue;
618
- const span = projectOtlpFlatLine(row);
619
- if (!span) continue;
620
- const arr = byTrace.get(span.trace_id);
621
- if (arr) arr.push(span);
622
- else byTrace.set(span.trace_id, [span]);
623
- }
624
- return byTrace;
625
- }
626
- function groupSpansByLogicalRun(byTrace, logicalRunIdForTrace) {
627
- if (!logicalRunIdForTrace) return byTrace;
628
- const byRun = /* @__PURE__ */ new Map();
629
- for (const [traceId, spans] of byTrace) {
630
- const suppliedRunId = logicalRunIdForTrace(traceId);
631
- if (typeof suppliedRunId !== "string" || suppliedRunId.trim().length === 0) throw new Error(`otlpToRunRecords: logicalRunIdForTrace('${traceId}') returned an empty run id`);
632
- const runId = suppliedRunId.trim();
633
- const target = byRun.get(runId) ?? [];
634
- for (const span of spans) target.push({
635
- ...span,
636
- span_id: qualifySpanId(traceId, span.span_id),
637
- parent_span_id: span.parent_span_id ? qualifySpanId(traceId, span.parent_span_id) : null
638
- });
639
- byRun.set(runId, target);
640
- }
641
- return byRun;
642
- }
643
- function qualifySpanId(traceId, spanId) {
644
- const prefix = `${traceId}:`;
645
- return spanId.startsWith(prefix) ? spanId : `${prefix}${spanId}`;
646
- }
647
- function aggregateTrace(traceId, spans, fallbackModel) {
648
- const ordered = [...spans].sort((a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id));
649
- const measurements = summarizeExecutionMeasurements(ordered.map((span) => ({
650
- id: span.span_id,
651
- ...span.parent_span_id ? { parentId: span.parent_span_id } : {},
652
- attributes: span.attributes,
653
- modelCall: isOtlpModelCall({
654
- kind: span.kind,
655
- name: span.name,
656
- attributes: span.attributes
657
- }),
658
- aggregate: span.kind !== "LLM" && span.kind !== "UNKNOWN"
659
- })));
660
- let toolSpanCount = 0;
661
- let agentSpanCount = 0;
662
- let firstErrorMessage;
663
- const modelVotes = /* @__PURE__ */ new Map();
664
- let earliest = ordered[0]?.start_time ?? "";
665
- let latest = ordered[0]?.end_time ?? "";
666
- for (const s of ordered) {
667
- if (s.start_time && (!earliest || compareSpanTime(s.start_time, earliest) < 0)) earliest = s.start_time;
668
- if (s.end_time && (!latest || compareSpanTime(s.end_time, latest) > 0)) latest = s.end_time;
669
- if (s.kind === "TOOL") toolSpanCount += 1;
670
- else if (s.kind === "AGENT") agentSpanCount += 1;
671
- if (s.status === "ERROR") {
672
- if (firstErrorMessage === void 0) firstErrorMessage = (s.status_message ?? `${s.name} — STATUS_CODE_ERROR`).slice(0, 500);
673
- }
674
- }
675
- const callSpanIds = new Set(measurements.callSpanIds);
676
- for (const span of ordered) {
677
- if (!callSpanIds.has(span.span_id)) continue;
678
- const model = firstStringAttr(span.attributes, LLM_MODEL_ATTR_KEYS) ?? span.model_name;
679
- if (model) modelVotes.set(model, (modelVotes.get(model) ?? 0) + 1);
680
- }
681
- const model = topVote(modelVotes) ?? firstModelAttr(ordered) ?? fallbackModel;
682
- let wallMs = 0;
683
- const a = spanEpochMillis(earliest);
684
- const b = spanEpochMillis(latest);
685
- if (a !== null && b !== null) wallMs = Math.max(0, b - a);
686
- const sourceTraceIds = [...new Set(spans.map((span) => span.trace_id))].sort();
687
- const terminal = terminalEvidenceFromRoots(ordered);
688
- const errorSummary = summarizeTraceErrors(ordered.map((span) => ({
689
- id: span.span_id,
690
- ...span.parent_span_id ? { parentId: span.parent_span_id } : {},
691
- role: errorRoleForProjectedSpan(span),
692
- error: span.status === "ERROR",
693
- processRoot: span.parent_span_id === null && isTerminalRootCandidate(span)
694
- })));
695
- return {
696
- traceId,
697
- sourceTraceCount: sourceTraceIds.length,
698
- sourceTraceIds,
699
- spanCount: spans.length,
700
- llmSpanCount: measurements.modelCallCount,
701
- toolSpanCount,
702
- agentSpanCount,
703
- errorSpanCount: errorSummary.total,
704
- executionErrorCount: errorSummary.execution,
705
- processErrorCount: errorSummary.process,
706
- guardrailErrorCount: errorSummary.guardrail,
707
- judgeErrorCount: errorSummary.evaluation,
708
- propagatedErrorCount: errorSummary.propagated,
709
- unclassifiedErrorCount: errorSummary.unclassified,
710
- tokenUsage: measurements.tokenUsage,
711
- firstErrorMessage,
712
- model,
713
- startTime: earliest,
714
- endTime: latest,
715
- wallMs,
716
- terminalOutcome: terminal.outcome,
717
- ...terminal.failureMessage ? { terminalFailureMessage: terminal.failureMessage } : {},
718
- callSpanIds: measurements.callSpanIds,
719
- costMeasurement: measurements.cost,
720
- ...measurements.aggregate ? { aggregateMeasurement: measurements.aggregate } : {}
721
- };
722
- }
723
- function terminalEvidenceFromRoots(spans) {
724
- const roots = spans.filter((span) => span.parent_span_id === null && isTerminalRootCandidate(span));
725
- if (roots.length !== 1) return { outcome: "unknown" };
726
- const root = roots[0];
727
- if (root.status === "ERROR") return {
728
- outcome: "failed",
729
- failureMessage: (root.status_message ?? `${root.name} — STATUS_CODE_ERROR`).slice(0, 500)
730
- };
731
- if (root.status === "OK") return { outcome: "succeeded" };
732
- return { outcome: "unknown" };
733
- }
734
- function isTerminalRootCandidate(span) {
735
- const role = errorRoleForProjectedSpan(span);
736
- return role !== "LLM" && role !== "TOOL" && role !== "EVALUATOR" && role !== "GUARDRAIL";
737
- }
738
- function errorRoleForProjectedSpan(span) {
739
- return classifyOtlpSpanRole({
740
- kind: span.kind,
741
- name: span.name,
742
- attributes: span.attributes
743
- });
744
- }
745
- function resolveScore(opts, traceId, agg) {
746
- const supplied = opts.scoreForTrace?.(traceId, agg);
747
- if (supplied !== void 0) {
748
- if (!Number.isFinite(supplied)) throw new Error(`otlpToRunRecords: scoreForTrace('${traceId}') returned non-finite ${supplied}`);
749
- return supplied;
750
- }
751
- }
752
- function resolveCost(opts, agg) {
753
- const observedCost = agg.costMeasurement;
754
- if (observedCost.complete && observedCost.value !== void 0) return {
755
- costUsd: observedCost.value,
756
- costProvenance: {
757
- kind: "observed",
758
- usd: observedCost.value
759
- }
760
- };
761
- if (agg.aggregateMeasurement?.costUsd !== void 0) return {
762
- costUsd: agg.aggregateMeasurement.costUsd,
763
- costProvenance: {
764
- kind: "observed",
765
- usd: agg.aggregateMeasurement.costUsd
766
- }
767
- };
768
- if (opts.priceUsdPerToken !== void 0) {
769
- const costUsd = (agg.tokenUsage.input + agg.tokenUsage.output) * opts.priceUsdPerToken;
770
- return {
771
- costUsd,
772
- costProvenance: {
773
- kind: "estimated",
774
- usd: costUsd
775
- }
776
- };
777
- }
778
- return {
779
- costUsd: null,
780
- costProvenance: {
781
- kind: "uncaptured",
782
- usd: null
783
- }
784
- };
785
- }
786
- function extractPromptCompletion(spans, callSpanIds) {
787
- const callIds = new Set(callSpanIds);
788
- const measuredCalls = spans.filter((span) => callIds.has(span.span_id));
789
- const llm = (measuredCalls.length > 0 ? measuredCalls : spans.filter((s) => s.kind === "LLM")).sort((a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id));
790
- if (llm.length === 0) return {};
791
- const promptText = firstStringAttr(llm[0].attributes, [
792
- "input.value",
793
- "llm.input_messages",
794
- "gen_ai.prompt"
795
- ]) ?? void 0;
796
- const last = llm[llm.length - 1];
797
- const completionText = firstStringAttr(last.attributes, [
798
- "output.value",
799
- "llm.output_messages",
800
- "gen_ai.completion"
801
- ]) ?? void 0;
802
- return {
803
- ...promptText !== void 0 ? { promptText } : {},
804
- ...completionText !== void 0 ? { completionText } : {}
805
- };
806
- }
807
- function topVote(votes) {
808
- let best = null;
809
- let bestN = 0;
810
- for (const [k, n] of votes) if (n > bestN || n === bestN && best !== null && k < best) {
811
- best = k;
812
- bestN = n;
813
- }
814
- return best;
815
- }
816
- function firstModelAttr(spans) {
817
- for (const s of spans) {
818
- const m = firstStringAttr(s.attributes, LLM_MODEL_ATTR_KEYS) ?? s.model_name;
819
- if (m) return m;
820
- }
821
- return null;
822
- }
823
- /**
824
- * `validateRunRecord` rejects bare model aliases (`gpt-4o`) that remap
825
- * silently. AppWorld/HALO traces frequently carry such bare ids (or a null
826
- * model). When the model already encodes a snapshot we keep it; otherwise we
827
- * append the fallback snapshot token so the row is admissible without lying
828
- * about the model — the bare base name is preserved verbatim before `@`.
829
- */
830
- function ensureSnapshot(model, fallbackModel) {
831
- if (modelHasSnapshot(model)) return model;
832
- return `${model}${fallbackModel.includes("@") ? fallbackModel.slice(fallbackModel.indexOf("@")) : "@otlp"}`;
833
- }
834
- //#endregion
835
- //#region src/trace-analyst/store-tool-spans.ts
836
- /** Missing tool spans cannot distinguish a tool-free run from broken capture. */
837
- var ToolTraceMissingError = class extends CaptureIntegrityError {
838
- constructor() {
839
- super("toolSpansToTraceAnalysisStore: no tool spans supplied; trace evidence is missing");
840
- }
841
- };
842
- /**
843
- * Snapshot canonical tool spans into the read interface used by trace analysts.
844
- * One run becomes one trace while arguments, results, errors, and timing remain searchable.
845
- */
846
- function toolSpansToTraceAnalysisStore(spans, options = {}) {
847
- if (!spans || spans.length === 0) throw new ToolTraceMissingError();
848
- const seen = /* @__PURE__ */ new Set();
849
- const lines = spans.map((span, index) => {
850
- assertToolSpanIdentity(span, index);
851
- const identity = `${span.runId}\u0000${span.spanId}`;
852
- if (seen.has(identity)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: duplicate span '${span.spanId}' in run '${span.runId}'`);
853
- seen.add(identity);
854
- const attributes = { ...span.attributes ?? {} };
855
- applyToolSpanOtlpAttributes(attributes, span);
856
- attributes[OPENINFERENCE_SPAN_KIND] = "TOOL";
857
- const endedAt = span.endedAt ?? span.startedAt + (span.latencyMs ?? 0);
858
- const line = createOtlpFlatLine({
859
- traceId: span.runId,
860
- spanId: span.spanId,
861
- parentSpanId: span.parentSpanId ?? null,
862
- name: span.name,
863
- kind: "SPAN_KIND_INTERNAL",
864
- startTime: toolSpanTimeIso(span.startedAt, span.spanId, "startedAt"),
865
- endTime: toolSpanTimeIso(endedAt, span.spanId, "endedAt"),
866
- statusCode: spanStatusToOtlp(span.status, span.error, "STATUS_CODE_UNSET"),
867
- statusMessage: span.error,
868
- resource: { attributes: {} },
869
- attributes
870
- });
871
- try {
872
- return JSON.stringify(line);
873
- } catch (cause) {
874
- throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' in run '${span.runId}' is not JSON-serializable`, { cause });
875
- }
876
- });
877
- return createOtlpBufferTraceStore(Buffer.from(`${lines.join("\n")}\n`, "utf8"), options);
878
- }
879
- function assertToolSpanIdentity(span, index) {
880
- if (span.kind !== "tool") throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span at index ${index} has kind '${String(span.kind)}', not 'tool'`);
881
- if (!span.runId || !span.spanId || !span.name || !span.toolName) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span at index ${index} is missing runId, spanId, name, or toolName`);
882
- if (!Number.isFinite(span.startedAt)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid startedAt`);
883
- if (span.endedAt !== void 0 && !Number.isFinite(span.endedAt)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid endedAt`);
884
- if (span.endedAt !== void 0 && span.endedAt < span.startedAt) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' ends before it starts`);
885
- if (span.latencyMs !== void 0 && (!Number.isFinite(span.latencyMs) || span.latencyMs < 0)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid latencyMs`);
886
- }
887
- function toolSpanTimeIso(value, spanId, field) {
888
- const iso = epochMillisToIso(value);
889
- if (iso) return iso;
890
- throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${spanId}' has invalid ${field}`);
891
- }
892
- //#endregion
893
- //#region src/trace/capture-fetch.ts
894
- /**
895
- * Wrap a provider `fetch` and record request, response, and error events.
896
- *
897
- * The returned value is a plain `typeof fetch`. Capture is best-effort by
898
- * default; set `failClosed` when telemetry loss must stop the provider call.
899
- */
900
- const DEFAULT_BODY_CAP = 2 * 1024 * 1024;
901
- function headersToRecord(headers) {
902
- if (!headers) return void 0;
903
- const out = {};
904
- headers.forEach((value, key) => {
905
- out[key.toLowerCase()] = value;
906
- });
907
- return Object.keys(out).length > 0 ? out : void 0;
908
- }
909
- function parseMaybeJson(text) {
910
- if (text.length === 0) return void 0;
911
- try {
912
- return JSON.parse(text);
913
- } catch {
914
- return text;
915
- }
916
- }
917
- /** Best-effort request-body read across the `fetch` input forms. */
918
- async function readRequestBody(input, init) {
919
- if (typeof init?.body === "string") return parseMaybeJson(init.body);
920
- if (init?.body != null) return void 0;
921
- if (input instanceof Request) try {
922
- return parseMaybeJson(await input.clone().text());
923
- } catch {
924
- return;
925
- }
926
- }
927
- function endpointFromUrl(url, baseUrl) {
928
- const normalisedBase = baseUrl.replace(/\/+$/, "");
929
- if (url.startsWith(normalisedBase)) return url.slice(normalisedBase.length) || "/";
930
- try {
931
- return new URL(url).pathname;
932
- } catch {
933
- return url;
934
- }
935
- }
936
- function captureFetchToRawSink(fetch, sink, ctx, opts = {}) {
937
- const provider = ctx.provider ?? providerFromBaseUrl(ctx.baseUrl);
938
- const redactor = opts.redactor ?? defaultProviderRedactor;
939
- const bodyCap = opts.responseBodyByteCap ?? DEFAULT_BODY_CAP;
940
- let warned = false;
941
- const baseEvent = (direction, endpoint) => ({
942
- eventId: crypto.randomUUID(),
943
- runId: ctx.runId,
944
- spanId: ctx.spanId,
945
- provider,
946
- model: ctx.model,
947
- endpoint,
948
- baseUrl: ctx.baseUrl,
949
- attemptIndex: 0,
950
- direction,
951
- timestamp: Date.now(),
952
- redactedFields: []
953
- });
954
- const record = async (event) => {
955
- try {
956
- await sink.record(redactor(event));
957
- } catch (err) {
958
- if (opts.failClosed) throw err;
959
- if (!warned) {
960
- warned = true;
961
- console.warn(`captureFetchToRawSink: sink.record failed (capture is best-effort) — ${err instanceof Error ? err.message : String(err)}`);
962
- }
963
- }
964
- };
965
- return async (input, init) => {
966
- const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
967
- const method = (init?.method ?? (input instanceof Request ? input.method : "GET")).toUpperCase();
968
- const endpoint = endpointFromUrl(url, ctx.baseUrl);
969
- const reqHeaders = new Headers(init?.headers ?? (input instanceof Request ? input.headers : void 0));
970
- await record({
971
- ...baseEvent("request", endpoint),
972
- requestHeaders: {
973
- ...headersToRecord(reqHeaders),
974
- "x-http-method": method
975
- },
976
- requestBody: await readRequestBody(input, init)
977
- });
978
- const start = Date.now();
979
- let response;
980
- try {
981
- response = await fetch(input, init);
982
- } catch (err) {
983
- await record({
984
- ...baseEvent("error", endpoint),
985
- durationMs: Date.now() - start,
986
- errorMessage: err instanceof Error ? err.message : String(err)
987
- });
988
- throw err;
989
- }
990
- let responseBody;
991
- let rawText;
992
- const redactedFields = [];
993
- try {
994
- rawText = await response.clone().text();
995
- if (rawText.length > bodyCap) {
996
- responseBody = rawText.slice(0, bodyCap);
997
- redactedFields.push("body_truncated");
998
- } else responseBody = parseMaybeJson(rawText);
999
- } catch {
1000
- responseBody = void 0;
1001
- }
1002
- if (opts.onUsage && rawText !== void 0) try {
1003
- const parsedForUsage = parseMaybeJson(rawText);
1004
- const usage = extractUsage(parsedForUsage) ?? (typeof parsedForUsage === "string" ? extractUsageFromSse(rawText, { mode: opts.sseUsageMode }) : null);
1005
- if (usage) opts.onUsage(usage, ctx);
1006
- } catch (err) {
1007
- if (opts.failClosed) throw err;
1008
- }
1009
- await record({
1010
- ...baseEvent("response", endpoint),
1011
- durationMs: Date.now() - start,
1012
- statusCode: response.status,
1013
- responseHeaders: headersToRecord(response.headers),
1014
- responseBody,
1015
- redactedFields
1016
- });
1017
- return response;
1018
- };
1019
- }
1020
- //#endregion
1021
- //#region src/trace/wire-ids.ts
1022
- /**
1023
- * The ONE mapping from agent-eval's human-readable ids (run ids, span labels)
1024
- * to W3C/OTLP wire ids. Every exporter in this package MUST route through
1025
- * these two functions — two exporters with private paddings once produced
1026
- * DIFFERENT trace ids for the same run, and one emitted invalid hex embedding
1027
- * the raw run id in the wire id (tangle-network/agent-runtime#694).
1028
- *
1029
- * Semantics:
1030
- * - an id that is ALREADY a valid W3C id passes through unchanged, so a
1031
- * trace id received from an inbound `traceparent` survives the round-trip
1032
- * and cross-process correlation is preserved;
1033
- * - anything else is derived with the contract's `deriveHexId`, the only
1034
- * legal derivation — deterministic, so every process that derives from the
1035
- * same human id mints the SAME wire id.
1036
- */
1037
- /** 32-hex W3C trace id for any id string. */
1038
- function traceIdForWire(id) {
1039
- return isW3CTraceId(id) ? id : deriveHexId(id, 16);
1040
- }
1041
- /** 16-hex W3C span id for any id string. */
1042
- function spanIdForWire(id) {
1043
- return isW3CSpanId(id) ? id : deriveHexId(id, 8);
1044
- }
1045
- //#endregion
1046
- //#region src/trace/otel.ts
1047
- /**
1048
- * OpenTelemetry JSON export — maps TraceSchema v1 to OTLP/JSON so
1049
- * traces render natively in Jaeger / Honeycomb / Langfuse / Grafana.
1050
- *
1051
- * Wire format only. We do NOT depend on the @opentelemetry SDK — that
1052
- * would drag in polyfills incompatible with Workers/Edge. Consumers
1053
- * push the JSON to their collector of choice via HTTP.
1054
- *
1055
- * Reference: OTLP 1.3.2 (ResourceSpans / ScopeSpans / Span).
1056
- */
1057
- const OTEL_AGENT_EVAL_SCOPE = {
1058
- name: "@tangle-network/agent-eval",
1059
- version: "0.3.0"
1060
- };
1061
- /** Export a single run's spans + events in OTLP/JSON. */
1062
- async function exportRunAsOtlp(store, runId, resourceAttrs = {}) {
1063
- const run = await store.getRun(runId);
1064
- if (!run) throw new Error(`run ${runId} not found`);
1065
- const spans = await store.spans({ runId });
1066
- const events = await store.events({ runId });
1067
- const eventsBySpan = /* @__PURE__ */ new Map();
1068
- for (const e of events) {
1069
- if (!e.spanId) continue;
1070
- const arr = eventsBySpan.get(e.spanId) ?? [];
1071
- arr.push(e);
1072
- eventsBySpan.set(e.spanId, arr);
1073
- }
1074
- const traceId = runToTraceId(run);
1075
- const otlpSpans = spans.map((s) => spanToOtlp(s, traceId, eventsBySpan.get(s.spanId) ?? []));
1076
- return { resourceSpans: [{
1077
- resource: { attributes: toAttributes$1({
1078
- "service.name": "agent-eval",
1079
- "run.id": run.runId,
1080
- "run.scenario_id": run.scenarioId,
1081
- "run.variant_id": run.variantId ?? "",
1082
- "run.dataset_version": run.datasetVersion ?? "",
1083
- "run.code_sha": run.codeSha ?? "",
1084
- "run.model_fingerprint": run.modelFingerprint ?? "",
1085
- ...resourceAttrs
1086
- }) },
1087
- scopeSpans: [{
1088
- scope: OTEL_AGENT_EVAL_SCOPE,
1089
- spans: otlpSpans
1090
- }]
1091
- }] };
1092
- }
1093
- function spanToOtlp(span, traceId, events) {
1094
- const endedAt = span.endedAt ?? span.startedAt;
1095
- return {
1096
- traceId,
1097
- spanId: spanIdForWire(span.spanId),
1098
- parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : void 0,
1099
- name: span.name,
1100
- kind: 1,
1101
- startTimeUnixNano: msToNs$1(span.startedAt),
1102
- endTimeUnixNano: msToNs$1(endedAt),
1103
- attributes: toAttributes$1(flattenSpanAttributes(span)),
1104
- events: events.map((e) => ({
1105
- timeUnixNano: msToNs$1(e.timestamp),
1106
- name: e.kind,
1107
- attributes: toAttributes$1(flattenPayload(e.payload))
1108
- })),
1109
- status: span.status === "error" ? {
1110
- code: 2,
1111
- message: span.error
1112
- } : { code: 1 }
1113
- };
1114
- }
1115
- function flattenSpanAttributes(span) {
1116
- const base = {};
1117
- if (span.attributes) {
1118
- for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") base[k] = v;
1119
- }
1120
- base[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
1121
- if (span.kind === "llm") applyLlmSpanOtlpAttributes(base, span);
1122
- else if (span.kind === "tool") applyToolSpanOtlpAttributes(base, span);
1123
- else if (span.kind === "retrieval") {
1124
- base["retrieval.query"] = span.query;
1125
- base["retrieval.hits"] = span.hits.length;
1126
- } else if (span.kind === "judge") {
1127
- base["judge.id"] = span.judgeId;
1128
- base["judge.dimension"] = span.dimension;
1129
- base["judge.score"] = span.score;
1130
- base["judge.target_span_id"] = span.targetSpanId;
1131
- } else if (span.kind === "sandbox") {
1132
- if (span.image) base["sandbox.image"] = span.image;
1133
- if (span.exitCode !== void 0) base["sandbox.exit_code"] = span.exitCode;
1134
- if (span.testsPassed !== void 0) base["sandbox.tests_passed"] = span.testsPassed;
1135
- if (span.testsTotal !== void 0) base["sandbox.tests_total"] = span.testsTotal;
1136
- }
1137
- return base;
1138
- }
1139
- function flattenPayload(payload) {
1140
- const out = {};
1141
- for (const [k, v] of Object.entries(payload)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") out[k] = v;
1142
- else out[k] = JSON.stringify(v);
1143
- return out;
1144
- }
1145
- function toAttributes$1(record) {
1146
- return Object.entries(record).map(([key, value]) => ({
1147
- key,
1148
- value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
1149
- }));
1150
- }
1151
- function msToNs$1(ms) {
1152
- return (BigInt(Math.floor(ms)) * 1000000n).toString();
1153
- }
1154
- function runToTraceId(run) {
1155
- return traceIdForWire(run.runId);
1156
- }
1157
- //#endregion
1158
- //#region src/trace/otel-bridge.ts
1159
- /**
1160
- * Create a RunCompleteHook that exports all spans from the completed run
1161
- * to the OTEL exporter, then flushes.
1162
- */
1163
- function otelRunCompleteHook(exporter) {
1164
- return async (ctx) => {
1165
- const spans = await ctx.store.spans({ runId: ctx.runId });
1166
- for (const span of spans) if (span.endedAt) exporter.exportSpan(storeSpanToExportable(span, ctx.runId));
1167
- await exporter.flush();
1168
- };
1169
- }
1170
- /**
1171
- * Create an auto-exporting TraceStore wrapper that intercepts updateSpan
1172
- * calls. When a span gets an endedAt, it's exported immediately. This
1173
- * gives real-time streaming instead of batch-at-end.
1174
- *
1175
- * This is the preferred integration path: wrap the store before
1176
- * constructing the TraceEmitter.
1177
- */
1178
- function createOtelTracingStore(inner, exporter, traceId) {
1179
- return {
1180
- async appendRun(run) {
1181
- return inner.appendRun(run);
1182
- },
1183
- async updateRun(runId, patch) {
1184
- return inner.updateRun(runId, patch);
1185
- },
1186
- async appendSpan(span) {
1187
- if (span.endedAt) exporter.exportSpan(storeSpanToExportable(span, traceId));
1188
- return inner.appendSpan(span);
1189
- },
1190
- async updateSpan(spanId, patch) {
1191
- await inner.updateSpan(spanId, patch);
1192
- if (patch.endedAt) {
1193
- const found = (await inner.spans({ runId: traceId })).find((s) => s.spanId === spanId);
1194
- if (found) exporter.exportSpan(storeSpanToExportable(found, traceId));
1195
- }
1196
- },
1197
- async appendEvent(event) {
1198
- return inner.appendEvent(event);
1199
- },
1200
- async appendBudgetEntry(entry) {
1201
- return inner.appendBudgetEntry(entry);
1202
- },
1203
- async appendArtifact(artifact) {
1204
- return inner.appendArtifact(artifact);
1205
- },
1206
- getRun: inner.getRun.bind(inner),
1207
- listRuns: inner.listRuns.bind(inner),
1208
- spans: inner.spans.bind(inner),
1209
- events: inner.events.bind(inner),
1210
- budget: inner.budget.bind(inner),
1211
- artifacts: inner.artifacts.bind(inner)
1212
- };
1213
- }
1214
- function storeSpanToExportable(span, traceId) {
1215
- const llm = span.kind === "llm" ? span : void 0;
1216
- const tool = span.kind === "tool" ? span : void 0;
1217
- return {
1218
- traceId,
1219
- spanId: span.spanId,
1220
- parentSpanId: span.parentSpanId,
1221
- name: span.name,
1222
- kind: span.kind,
1223
- startedAt: span.startedAt,
1224
- endedAt: span.endedAt,
1225
- status: span.status,
1226
- error: span.error,
1227
- model: llm?.model,
1228
- inputTokens: llm?.inputTokens,
1229
- outputTokens: llm?.outputTokens,
1230
- reasoningTokens: llm?.reasoningTokens,
1231
- cachedTokens: llm?.cachedTokens,
1232
- cacheWriteTokens: llm?.cacheWriteTokens,
1233
- costUsd: llm?.costUsd,
1234
- tool: tool ? {
1235
- toolName: tool.toolName,
1236
- args: tool.args,
1237
- argsCaptured: tool.argsCaptured,
1238
- result: tool.result,
1239
- latencyMs: tool.latencyMs
1240
- } : void 0,
1241
- attributes: span.attributes
1242
- };
1243
- }
1244
- //#endregion
1245
- //#region src/trace/otel-export.ts
1246
- /**
1247
- * OTEL span exporter — streams spans to an OTLP/HTTP collector.
1248
- *
1249
- * Reads OTEL_EXPORTER_OTLP_ENDPOINT + OTEL_EXPORTER_OTLP_HEADERS from env
1250
- * when no explicit config is given. Batches spans and flushes periodically
1251
- * or when the batch fills. No @opentelemetry SDK dependency — minimal
1252
- * OTLP/JSON serializer (~120 LOC) using the existing otel.ts helpers.
1253
- */
1254
- /**
1255
- * Create an OTEL exporter. Returns undefined when no endpoint is configured
1256
- * (neither via config nor env) — callers should check before attaching.
1257
- */
1258
- function createOtelExporter(config) {
1259
- const resolvedEndpoint = config?.endpoint ?? (typeof process !== "undefined" ? process.env.OTEL_EXPORTER_OTLP_ENDPOINT : void 0);
1260
- if (!resolvedEndpoint) return void 0;
1261
- const endpoint = resolvedEndpoint;
1262
- const headers = config?.headers ?? parseHeadersFromEnv();
1263
- const batchSize = config?.batchSize ?? 64;
1264
- const flushIntervalMs = config?.flushIntervalMs ?? 5e3;
1265
- const serviceName = config?.serviceName ?? "agent-eval";
1266
- const resourceAttrs = config?.resourceAttributes ?? {};
1267
- const pending = [];
1268
- let timer;
1269
- let stopped = false;
1270
- const exporter = {
1271
- exportSpan(span) {
1272
- if (stopped) return;
1273
- pending.push(toOtlpSpan(span));
1274
- if (pending.length >= batchSize) doFlush();
1275
- },
1276
- async flush() {
1277
- await doFlush();
1278
- },
1279
- async shutdown() {
1280
- stopped = true;
1281
- if (timer !== void 0) {
1282
- clearInterval(timer);
1283
- timer = void 0;
1284
- }
1285
- await doFlush();
1286
- }
1287
- };
1288
- timer = setInterval(() => {
1289
- if (pending.length > 0) doFlush();
1290
- }, flushIntervalMs);
1291
- if (typeof timer === "object" && "unref" in timer) timer.unref();
1292
- async function doFlush() {
1293
- if (pending.length === 0) return;
1294
- const batch = pending.splice(0);
1295
- const body = { resourceSpans: [{
1296
- resource: { attributes: toAttributes({
1297
- "service.name": serviceName,
1298
- ...resourceAttrs
1299
- }) },
1300
- scopeSpans: [{
1301
- scope: OTEL_AGENT_EVAL_SCOPE,
1302
- spans: batch
1303
- }]
1304
- }] };
1305
- const url = `${endpoint.replace(/\/+$/, "")}/v1/traces`;
1306
- try {
1307
- await fetch(url, {
1308
- method: "POST",
1309
- headers: {
1310
- "content-type": "application/json",
1311
- ...headers
1312
- },
1313
- body: JSON.stringify(body)
1314
- });
1315
- } catch {}
1316
- }
1317
- return exporter;
1318
- }
1319
- function parseHeadersFromEnv() {
1320
- if (typeof process === "undefined") return {};
1321
- const raw = process.env.OTEL_EXPORTER_OTLP_HEADERS;
1322
- if (!raw) return {};
1323
- const out = {};
1324
- for (const pair of raw.split(",")) {
1325
- const eq = pair.indexOf("=");
1326
- if (eq < 0) continue;
1327
- const key = pair.slice(0, eq).trim();
1328
- const value = pair.slice(eq + 1).trim();
1329
- if (key) out[key] = value;
1330
- }
1331
- return out;
1332
- }
1333
- function toOtlpSpan(span) {
1334
- const endedAt = span.endedAt ?? span.startedAt;
1335
- const attrs = {};
1336
- if (span.attributes) {
1337
- for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") attrs[k] = v;
1338
- }
1339
- attrs[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
1340
- applyLlmSpanOtlpAttributes(attrs, span);
1341
- if (span.tool) applyToolSpanOtlpAttributes(attrs, span.tool);
1342
- return {
1343
- traceId: traceIdForWire(span.traceId),
1344
- spanId: spanIdForWire(span.spanId),
1345
- parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : void 0,
1346
- name: span.name,
1347
- kind: 1,
1348
- startTimeUnixNano: msToNs(span.startedAt),
1349
- endTimeUnixNano: msToNs(endedAt),
1350
- attributes: toAttributes(attrs),
1351
- status: span.status === "error" ? {
1352
- code: 2,
1353
- message: span.error
1354
- } : { code: 1 }
1355
- };
1356
- }
1357
- function toAttributes(record) {
1358
- return Object.entries(record).map(([key, value]) => ({
1359
- key,
1360
- value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
1361
- }));
1362
- }
1363
- function msToNs(ms) {
1364
- return (BigInt(Math.floor(ms)) * 1000000n).toString();
1365
- }
1366
- //#endregion
1367
- //#region src/trace/store-to-otlp.ts
1368
- /**
1369
- * Convert agent-eval's internal trace shape (`FileSystemTraceStore` → `Run`,
1370
- * `Span`, `TraceEvent`) into the OTLP-flat JSONL the trace analyst
1371
- * (`analyzeTraces` + `OtlpFileTraceStore`) reads.
1372
- *
1373
- * Eval harnesses shard a `FileSystemTraceStore` per cell (persona / variant)
1374
- * under a run directory. The analyst consumes a single OTLP-NDJSON file keyed
1375
- * on `trace_id` + `span_id` with `start_time`/`end_time` in ISO-8601 and
1376
- * resource + `attributes` rolled up per-span. This module walks every shard,
1377
- * projects each `Span` (plus events pinned to it) into the flat OTLP shape,
1378
- * and emits one NDJSON file.
1379
- *
1380
- * Generic OTLP/OpenInference fields are always emitted (`service.name`,
1381
- * `agent.name`, `run.id`/`run.status`, `openinference.span.kind`,
1382
- * `llm.model_name`, …). Domain attributes (`legal.*`, `tax.*`, …) are injected
1383
- * per-run via {@link TraceStoreToOtlpOptions.resourceAttributes} /
1384
- * {@link TraceStoreToOtlpOptions.runAttributes} so consumers don't re-roll the
1385
- * walker.
1386
- */
1387
- /**
1388
- * Read every per-cell shard under each source root and write a flat OTLP-JSONL
1389
- * view of the corpus to `outPath`. Each cell directory is a
1390
- * `FileSystemTraceStore` — NDJSON append-only with size-based rotation;
1391
- * `updateRun`/`updateSpan` append `{ id, ...patch, _update: true }` rows
1392
- * rather than rewriting, so readers must merge those patches in (done here).
1393
- *
1394
- * A `string` source is treated as a celled root.
1395
- */
1396
- function convertTraceStoresToOtlp(source, outPath, opts = {}) {
1397
- const sources = Array.isArray(source) ? [...source] : typeof source === "string" ? [{
1398
- root: source,
1399
- layout: "celled"
1400
- }] : [source];
1401
- const defaultServiceName = opts.serviceName ?? "agent-eval";
1402
- const resourceAttributes = opts.resourceAttributes ?? (() => ({}));
1403
- const runAttributes = opts.runAttributes ?? (() => ({}));
1404
- const lines = [];
1405
- let spanCount = 0;
1406
- let runCount = 0;
1407
- let cellCount = 0;
1408
- let cellErrorCount = 0;
1409
- for (const src of sources) {
1410
- const serviceName = src.serviceName ?? defaultServiceName;
1411
- const cellDirs = src.layout === "flat" ? [{
1412
- label: "<root>",
1413
- dir: src.root
1414
- }] : listCells(src.root).map((name) => ({
1415
- label: name,
1416
- dir: join(src.root, name)
1417
- }));
1418
- for (const cell of cellDirs) try {
1419
- const result = projectCell({
1420
- cellDir: cell.dir,
1421
- serviceName,
1422
- resourceAttributes,
1423
- runAttributes
1424
- });
1425
- for (const line of result.lines) lines.push(line);
1426
- spanCount += result.spanCount;
1427
- runCount += result.runCount;
1428
- cellCount += 1;
1429
- } catch (err) {
1430
- console.warn(`[traces-to-otlp] cell ${cell.label} (${cell.dir}) skipped: ${err instanceof Error ? err.message : String(err)}`);
1431
- cellErrorCount += 1;
1432
- }
1433
- }
1434
- writeFileSync(outPath, lines.join("\n") + (lines.length > 0 ? "\n" : ""));
1435
- return {
1436
- spanCount,
1437
- runCount,
1438
- cellCount,
1439
- cellErrorCount
1440
- };
1441
- }
1442
- function projectCell(args) {
1443
- const { cellDir, serviceName, resourceAttributes, runAttributes } = args;
1444
- const lines = [];
1445
- let runCount = 0;
1446
- let spanCount = 0;
1447
- const runs = readMergedShards(cellDir, "runs", "runId");
1448
- const spans = readMergedShards(cellDir, "spans", "spanId");
1449
- const events = readShards(cellDir, "events");
1450
- const runByRunId = /* @__PURE__ */ new Map();
1451
- for (const r of runs) runByRunId.set(r.runId, r);
1452
- const spanBySpanId = /* @__PURE__ */ new Map();
1453
- for (const s of spans) spanBySpanId.set(s.spanId, s);
1454
- const eventsBySpanId = /* @__PURE__ */ new Map();
1455
- for (const e of events) {
1456
- if (e.kind === "state_mutation" && e.payload && typeof e.payload === "object") {
1457
- const entity = e.payload.entity;
1458
- if (entity === "run") {
1459
- const run = e.payload.run;
1460
- if (run?.runId) runByRunId.set(run.runId, run);
1461
- continue;
1462
- }
1463
- if (entity === "run.update") {
1464
- const patch = e.payload.patch;
1465
- if (patch && e.runId) {
1466
- const prior = runByRunId.get(e.runId);
1467
- if (prior) runByRunId.set(e.runId, {
1468
- ...prior,
1469
- ...patch
1470
- });
1471
- }
1472
- continue;
1473
- }
1474
- if (entity === "span") {
1475
- const span = e.payload.span;
1476
- if (span?.spanId) spanBySpanId.set(span.spanId, span);
1477
- continue;
1478
- }
1479
- if (entity === "span.update") {
1480
- const spanId = e.payload.spanId;
1481
- const patch = e.payload.patch;
1482
- if (spanId && patch) {
1483
- const prior = spanBySpanId.get(spanId);
1484
- if (prior) spanBySpanId.set(spanId, {
1485
- ...prior,
1486
- ...patch
1487
- });
1488
- }
1489
- continue;
1490
- }
1491
- }
1492
- if (!e.spanId) continue;
1493
- const arr = eventsBySpanId.get(e.spanId) ?? [];
1494
- arr.push(e);
1495
- eventsBySpanId.set(e.spanId, arr);
1496
- }
1497
- for (const run of runByRunId.values()) {
1498
- const traceId = traceIdForWire(run.runId);
1499
- const agentName = run.variantId ?? run.scenarioId;
1500
- const sharedResource = { attributes: {
1501
- "service.name": serviceName,
1502
- "agent.name": agentName,
1503
- "run.id": run.runId,
1504
- "run.status": run.status,
1505
- ...resourceAttributes(run)
1506
- } };
1507
- const runSpanId = spanIdForWire(`run-${run.runId}`);
1508
- const runStart = msToIso(run.startedAt);
1509
- const runEnd = msToIso(run.endedAt ?? run.startedAt);
1510
- const runStatus = run.outcome?.failureClass && run.outcome.failureClass !== "success" ? "STATUS_CODE_ERROR" : "STATUS_CODE_OK";
1511
- const runAttrs = {
1512
- [OPENINFERENCE_SPAN_KIND]: "AGENT",
1513
- "agent.name": agentName,
1514
- "agent.workflow.name": serviceName,
1515
- ...runAttributes(run)
1516
- };
1517
- lines.push(JSON.stringify(toLine({
1518
- traceId,
1519
- spanId: runSpanId,
1520
- parentSpanId: "",
1521
- name: `run.${agentName}`,
1522
- kind: "SPAN_KIND_INTERNAL",
1523
- startTime: runStart,
1524
- endTime: runEnd,
1525
- statusCode: runStatus,
1526
- statusMessage: run.outcome?.notes ?? "",
1527
- resource: sharedResource,
1528
- attributes: runAttrs
1529
- })));
1530
- runCount += 1;
1531
- for (const span of spanBySpanId.values()) {
1532
- if (span.runId !== run.runId) continue;
1533
- const spanAttrs = spanToAttributes(span, eventsBySpanId.get(span.spanId) ?? []);
1534
- const statusCode = spanStatusToOtlp(span.status, span.error, "STATUS_CODE_OK");
1535
- lines.push(JSON.stringify(toLine({
1536
- traceId,
1537
- spanId: spanIdForWire(span.spanId),
1538
- parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : runSpanId,
1539
- name: span.name,
1540
- kind: spanKindToOtlpKind(span.kind),
1541
- startTime: msToIso(span.startedAt),
1542
- endTime: msToIso(span.endedAt ?? span.startedAt),
1543
- statusCode,
1544
- statusMessage: span.error ?? "",
1545
- resource: sharedResource,
1546
- attributes: spanAttrs
1547
- })));
1548
- spanCount += 1;
1549
- }
1550
- }
1551
- return {
1552
- lines,
1553
- runCount,
1554
- spanCount
1555
- };
1556
- }
1557
- function listCells(root) {
1558
- try {
1559
- return readdirSync(root, { withFileTypes: true }).filter((d) => d.isDirectory()).map((d) => d.name).sort();
1560
- } catch {
1561
- return [];
1562
- }
1563
- }
1564
- /**
1565
- * Read every NDJSON shard for `name` under `cellDir`, ordered by mtime so
1566
- * rotated files apply before the active one. Yields raw rows including any
1567
- * `_update: true` patches.
1568
- */
1569
- function readShards(cellDir, name) {
1570
- let entries;
1571
- try {
1572
- entries = readdirSync(cellDir);
1573
- } catch {
1574
- return [];
1575
- }
1576
- const shards = entries.filter((f) => (f === `${name}.ndjson` || f.startsWith(`${name}.`)) && f.endsWith(".ndjson")).map((f) => ({
1577
- file: f,
1578
- path: join(cellDir, f)
1579
- })).map((s) => {
1580
- let mtime = 0;
1581
- try {
1582
- mtime = statSync(s.path).mtimeMs;
1583
- } catch {}
1584
- return {
1585
- ...s,
1586
- mtime
1587
- };
1588
- }).sort((a, b) => a.mtime - b.mtime || a.file.localeCompare(b.file));
1589
- const rows = [];
1590
- for (const shard of shards) {
1591
- let text;
1592
- try {
1593
- text = readFileSync(shard.path, "utf-8");
1594
- } catch {
1595
- continue;
1596
- }
1597
- for (const line of text.split("\n")) {
1598
- const trimmed = line.trim();
1599
- if (!trimmed) continue;
1600
- try {
1601
- rows.push(JSON.parse(trimmed));
1602
- } catch {}
1603
- }
1604
- }
1605
- return rows;
1606
- }
1607
- /**
1608
- * Read NDJSON shards and merge `{ ...patch, _update: true }` rows into the
1609
- * prior record keyed on `idKey` — mirrors the in-memory merge
1610
- * `FileSystemTraceStore` keeps but doesn't replay on cross-process load.
1611
- */
1612
- function readMergedShards(cellDir, name, idKey) {
1613
- const rows = readShards(cellDir, name);
1614
- const byId = /* @__PURE__ */ new Map();
1615
- for (const row of rows) {
1616
- const id = row[idKey];
1617
- if (!id) continue;
1618
- const prior = byId.get(id);
1619
- if (prior && row._update) byId.set(id, {
1620
- ...prior,
1621
- ...row,
1622
- _update: void 0
1623
- });
1624
- else byId.set(id, row);
1625
- }
1626
- return [...byId.values()];
1627
- }
1628
- function spanToAttributes(span, events) {
1629
- const attrs = { [OPENINFERENCE_SPAN_KIND]: traceSpanKindToOpenInferenceKind(span.kind) };
1630
- if (span.kind === "llm") {
1631
- applyLlmSpanOtlpAttributes(attrs, span);
1632
- if (Array.isArray(span.messages)) attrs["llm.input_messages"] = JSON.stringify(span.messages.slice(-6));
1633
- if (typeof span.output === "string") attrs["llm.output_messages"] = JSON.stringify([{
1634
- role: "assistant",
1635
- content: span.output
1636
- }]);
1637
- } else if (span.kind === "tool") applyToolSpanOtlpAttributes(attrs, span);
1638
- else if (span.kind === "judge") {
1639
- attrs["judge.id"] = span.judgeId;
1640
- attrs["judge.dimension"] = span.dimension;
1641
- attrs["judge.score"] = span.score;
1642
- attrs["judge.target_span_id"] = span.targetSpanId;
1643
- }
1644
- if (span.attributes) for (const [k, v] of Object.entries(span.attributes)) attrs[`agent_eval.${k}`] = v;
1645
- if (events.length > 0) {
1646
- attrs["agent_eval.event_count"] = events.length;
1647
- attrs["agent_eval.event_kinds"] = JSON.stringify(events.map((e) => e.kind));
1648
- }
1649
- return attrs;
1650
- }
1651
- function spanKindToOtlpKind(kind) {
1652
- switch (kind) {
1653
- case "llm": return "SPAN_KIND_CLIENT";
1654
- case "retrieval": return "SPAN_KIND_CLIENT";
1655
- default: return "SPAN_KIND_INTERNAL";
1656
- }
1657
- }
1658
- const toLine = createOtlpFlatLine;
1659
- function msToIso(ms) {
1660
- if (ms <= 0) return (/* @__PURE__ */ new Date(0)).toISOString();
1661
- return epochMillisToIso(ms) ?? (/* @__PURE__ */ new Date(0)).toISOString();
1662
- }
1663
- //#endregion
1664
- //#region src/replay.ts
1665
- /**
1666
- * Replay-from-raw-events — turn every captured campaign run into a
1667
- * re-runnable artifact.
1668
- *
1669
- * `RawProviderSink` captures every provider HTTP envelope; `runEvalCampaign`
1670
- * makes that capture the default. Together they make every past run a
1671
- * complete fingerprint of what happened on the wire — enough to replay
1672
- * the run without burning new LLM cost.
1673
- *
1674
- * Three use cases this primitive enables:
1675
- *
1676
- * 1. **Post-hoc judging** — apply a new judge / rubric / scoring callback
1677
- * to last week's runs without re-calling any LLM. The cost of trying
1678
- * a new rubric drops from "another full sweep" to a CPU-bound replay.
1679
- * 2. **Determinism audits** — replay the same campaign and verify the
1680
- * raw responses match byte-for-byte. Any drift is a non-determinism
1681
- * bug (in the harness, the prompt builder, the sandbox, …).
1682
- * 3. **Free judge calibration** — run two judges on identical responses
1683
- * and measure inter-judge agreement without doubling LLM spend.
1684
- *
1685
- * The interface is deliberately fetch-shaped. Inject `createReplayFetch`
1686
- * into `LlmClientOptions.fetch` and every `callLlm` transparently reads
1687
- * from the cache instead of calling the network. No new code path through
1688
- * the LLM client is needed; the cache hit is invisible to the runner.
1689
- */
1690
- var ReplayCacheMissError = class extends ReplayError {
1691
- url;
1692
- requestKey;
1693
- constructor(url, requestKey, message) {
1694
- super(message ?? `replay cache miss for ${url} (key=${requestKey})`);
1695
- this.url = url;
1696
- this.requestKey = requestKey;
1697
- }
1698
- };
1699
- /**
1700
- * In-memory deterministic cache of (request → response) keyed on a stable
1701
- * hash of the request body. Built from a `RawProviderSink` containing
1702
- * paired `request` and `response` events from a previous run.
1703
- *
1704
- * The cache is the source of truth for replay; `createReplayFetch` is a
1705
- * thin wrapper that reads from it.
1706
- */
1707
- var ReplayCache = class ReplayCache {
1708
- byKey = /* @__PURE__ */ new Map();
1709
- orphans = 0;
1710
- byProvider = {};
1711
- byModel = {};
1712
- /**
1713
- * Build a cache from a sink's events. The sink must implement `list()`.
1714
- * Filter by `runId` / `spanId` to scope to a specific replay.
1715
- */
1716
- static async fromSink(sink, filter = {}) {
1717
- if (!sink.list) throw new ReplayError("ReplayCache.fromSink: sink must implement list() to be replayable.");
1718
- const events = await sink.list(filter);
1719
- return ReplayCache.fromEvents(events);
1720
- }
1721
- /** Build a cache from an in-memory event list. */
1722
- static async fromEvents(events) {
1723
- const cache = new ReplayCache();
1724
- const groups = /* @__PURE__ */ new Map();
1725
- for (const e of events) {
1726
- const k = `${e.runId ?? ""}::${e.spanId ?? ""}::${e.attemptIndex}`;
1727
- const g = groups.get(k) ?? {};
1728
- if (e.direction === "request") g.req = e;
1729
- else g.res = e;
1730
- groups.set(k, g);
1731
- }
1732
- for (const g of groups.values()) {
1733
- if (!g.req) continue;
1734
- if (!g.res) {
1735
- cache.orphans += 1;
1736
- continue;
1737
- }
1738
- const key = await requestKey(g.req);
1739
- cache.byKey.set(key, {
1740
- request: g.req,
1741
- response: g.res
1742
- });
1743
- cache.byProvider[g.req.provider] = (cache.byProvider[g.req.provider] ?? 0) + 1;
1744
- cache.byModel[g.req.model] = (cache.byModel[g.req.model] ?? 0) + 1;
1745
- }
1746
- return cache;
1747
- }
1748
- /** Number of cacheable (request, response) pairs in the cache. */
1749
- size() {
1750
- return this.byKey.size;
1751
- }
1752
- stats() {
1753
- return {
1754
- total: this.byKey.size,
1755
- byProvider: { ...this.byProvider },
1756
- byModel: { ...this.byModel },
1757
- orphanRequests: this.orphans
1758
- };
1759
- }
1760
- /** Iterate every cached `(request, response)` pair in insertion order. */
1761
- *entries() {
1762
- for (const entry of this.byKey.values()) yield entry;
1763
- }
1764
- /**
1765
- * Look up a cached response by hashing the (model, messages, temperature,
1766
- * maxTokens, response_format) shape. Returns `undefined` on miss; the
1767
- * caller decides whether to throw, fall back to the network, or skip.
1768
- */
1769
- async lookup(requestBody) {
1770
- const key = await keyFromBody(requestBody);
1771
- return this.byKey.get(key);
1772
- }
1773
- };
1774
- /**
1775
- * Build a `fetch`-shaped function that serves cached responses out of a
1776
- * `ReplayCache` for any URL ending in `/chat/completions`. Pass through
1777
- * `LlmClientOptions.fetch` and `callLlm` becomes free.
1778
- *
1779
- * Non-`/chat/completions` URLs are passed straight to the fallback fetch
1780
- * (default: `globalThis.fetch`). This matters because non-LLM HTTP work
1781
- * (judge HTTP servers, sandbox callbacks) sometimes flows through the same
1782
- * `fetch` and shouldn't be intercepted.
1783
- */
1784
- function createReplayFetch(cache, opts = {}) {
1785
- const onMiss = opts.onMiss ?? "throw";
1786
- const fallback = opts.fallbackFetch ?? globalThis.fetch?.bind(globalThis);
1787
- return (async (input, init) => {
1788
- const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
1789
- if (!/\/chat\/completions(?:[?#].*)?$/.test(url)) {
1790
- if (!fallback) throw new ReplayError(`replay fetch: non-completions URL ${url} but no fallbackFetch configured`);
1791
- return fallback(input, init);
1792
- }
1793
- let bodyParsed;
1794
- if (init?.body && typeof init.body === "string") try {
1795
- bodyParsed = JSON.parse(init.body);
1796
- } catch {}
1797
- const hit = bodyParsed === void 0 ? void 0 : await cache.lookup(bodyParsed);
1798
- if (hit) {
1799
- opts.onHit?.({
1800
- url,
1801
- provider: hit.request.provider,
1802
- model: hit.request.model
1803
- });
1804
- const status = hit.response.statusCode ?? 200;
1805
- const headers = new Headers(Object.entries(hit.response.responseHeaders ?? { "Content-Type": "application/json" }));
1806
- const bodyText = typeof hit.response.responseBody === "string" ? hit.response.responseBody : JSON.stringify(hit.response.responseBody ?? {});
1807
- return new Response(bodyText, {
1808
- status,
1809
- headers
1810
- });
1811
- }
1812
- opts.onMissNotify?.({
1813
- url,
1814
- requestBody: bodyParsed
1815
- });
1816
- if (onMiss === "throw") throw new ReplayCacheMissError(url, bodyParsed === void 0 ? "<unparseable>" : await keyFromBody(bodyParsed));
1817
- if (onMiss === "fail-closed") return new Response(JSON.stringify({ error: "replay_cache_miss" }), { status: 599 });
1818
- if (!fallback) throw new ReplayError("replay fetch: onMiss=fallback but no fallbackFetch configured");
1819
- return fallback(input, init);
1820
- });
1821
- }
1822
- /**
1823
- * Convenience iterator over `(request, response)` pairs in a sink — for
1824
- * post-hoc scoring that doesn't need a `fetch` shim. The judge or scorer
1825
- * runs purely in-process over cached LLM outputs.
1826
- */
1827
- async function* iterateRawCalls(sink, filter = {}) {
1828
- if (!sink.list) throw new ReplayError("iterateRawCalls: sink must implement list().");
1829
- const events = await sink.list(filter);
1830
- const cache = await ReplayCache.fromEvents(events);
1831
- for (const entry of cache.entries()) yield entry;
1832
- }
1833
- /**
1834
- * Canonical request key.
1835
- *
1836
- * `model + messages + temperature + max_tokens|max_completion_tokens +
1837
- * response_format` are the dimensions that affect the response shape.
1838
- * Other fields (timestamp headers, provider-specific metadata) are
1839
- * intentionally excluded so a request hashes the same across re-runs.
1840
- */
1841
- async function requestKey(event) {
1842
- return keyFromBody(event.requestBody);
1843
- }
1844
- async function keyFromBody(body) {
1845
- if (body == null || typeof body !== "object") return hashJson({ raw: String(body) });
1846
- const b = body;
1847
- return hashJson(canonicalize({
1848
- model: b.model ?? null,
1849
- messages: b.messages ?? null,
1850
- temperature: b.temperature ?? null,
1851
- max_tokens: b.max_tokens ?? null,
1852
- max_completion_tokens: b.max_completion_tokens ?? null,
1853
- response_format: b.response_format ?? null
1854
- }));
1855
- }
1856
- //#endregion
1857
- export { TRACE_ANALYST_ACTOR_DESCRIPTION as A, domainEvidencePattern as C, tokenizeDomainWords as D, scoreTraceInsightReadiness as E, traceAnalystOnRunComplete as O, describeTraceInsightScope as S, planTraceInsightQuestions as T, otlpToTraceRunRecords as _, convertTraceStoresToOtlp as a, buildTraceInsightPrompt as b, otelRunCompleteHook as c, captureFetchToRawSink as d, ToolTraceMissingError as f, otlpToRunRecords as g, otlpRowsToTraceRunRecords as h, iterateRawCalls as i, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION as j, analyzeTraces as k, OTEL_AGENT_EVAL_SCOPE as l, otlpRowsToRunRecords as m, ReplayCacheMissError as n, createOtelExporter as o, toolSpansToTraceAnalysisStore as p, createReplayFetch as r, createOtelTracingStore as s, ReplayCache as t, exportRunAsOtlp as u, flattenOtlpExportToNdjson as v, inferDomainKeywords as w, defaultTraceInsightPanel as x, buildTraceInsightContext as y };
1858
-
1859
- //# sourceMappingURL=replay-CohS93nE.js.map