@tangle-network/agent-eval 0.144.11 → 0.144.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (378) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
  3. package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
  4. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
  5. package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
  6. package/dist/analyst/index.d.ts +134 -16
  7. package/dist/analyst/index.d.ts.map +1 -1
  8. package/dist/analyst/index.js +364 -10
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
  11. package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
  12. package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
  13. package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
  14. package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
  15. package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
  16. package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
  17. package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
  18. package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
  19. package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
  20. package/dist/benchmarks/index.d.ts +244 -2
  21. package/dist/benchmarks/index.d.ts.map +1 -0
  22. package/dist/benchmarks/index.js +733 -1
  23. package/dist/benchmarks/index.js.map +1 -0
  24. package/dist/builder-eval/index.d.ts +23 -2
  25. package/dist/builder-eval/index.d.ts.map +1 -1
  26. package/dist/builder-eval/index.js +227 -3
  27. package/dist/builder-eval/index.js.map +1 -1
  28. package/dist/campaign/index.d.ts +10 -8
  29. package/dist/campaign/index.js +9 -6
  30. package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
  31. package/dist/campaign-BYjBAypg.js.map +1 -0
  32. package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
  33. package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
  34. package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
  35. package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
  36. package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
  37. package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
  38. package/dist/cli.js +2 -2
  39. package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
  40. package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
  41. package/dist/contract/index.d.ts +13 -390
  42. package/dist/contract/index.d.ts.map +1 -1
  43. package/dist/contract/index.js +18 -542
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
  46. package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
  47. package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
  48. package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
  49. package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
  50. package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
  51. package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
  52. package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
  53. package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
  54. package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
  55. package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
  56. package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
  57. package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
  58. package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
  59. package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
  60. package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
  61. package/dist/descriptive-B5MwKfbf.js +144 -0
  62. package/dist/descriptive-B5MwKfbf.js.map +1 -0
  63. package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
  64. package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
  65. package/dist/effect-sizes-DiH8MGOH.js +82 -0
  66. package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
  67. package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
  68. package/dist/engine-otFpE2gF.d.ts.map +1 -0
  69. package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
  70. package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
  71. package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
  72. package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
  73. package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
  74. package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
  75. package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
  76. package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
  77. package/dist/experiment/index.d.ts +9 -6
  78. package/dist/experiment/index.d.ts.map +1 -1
  79. package/dist/experiment/index.js +11 -7
  80. package/dist/experiment/index.js.map +1 -1
  81. package/dist/experiment-tracker-C29gXM4B.js +269 -0
  82. package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
  83. package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
  84. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
  85. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
  86. package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
  87. package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
  88. package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
  89. package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
  90. package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
  91. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  92. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  93. package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
  94. package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
  95. package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
  96. package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
  97. package/dist/fuzz.d.ts +2 -2
  98. package/dist/fuzz.js +2 -2
  99. package/dist/hosted/index.d.ts +3 -3
  100. package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
  101. package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
  102. package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
  103. package/dist/index-BWDrSVfw.d.ts.map +1 -0
  104. package/dist/index-Ba3YrbAL.d.ts +1 -0
  105. package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
  106. package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
  107. package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
  108. package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
  109. package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
  110. package/dist/index-DSmEylT9.d.ts.map +1 -0
  111. package/dist/index.d.ts +2397 -5308
  112. package/dist/index.d.ts.map +1 -1
  113. package/dist/index.js +5914 -10496
  114. package/dist/index.js.map +1 -1
  115. package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
  116. package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
  117. package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
  118. package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
  119. package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
  120. package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
  121. package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
  122. package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
  123. package/dist/internal-BDHPCnjk.js +230 -0
  124. package/dist/internal-BDHPCnjk.js.map +1 -0
  125. package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
  126. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
  127. package/dist/judge-calibration-DZkWrm5H.js +317 -0
  128. package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
  129. package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
  130. package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
  131. package/dist/ledger-core/index.d.ts +1 -1
  132. package/dist/ledger-core/index.js +2 -2
  133. package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
  134. package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
  135. package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
  136. package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
  137. package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
  138. package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
  139. package/dist/matrix/index.d.ts +2 -2
  140. package/dist/meta-eval/index.d.ts +3 -3
  141. package/dist/meta-eval/index.js +3 -3
  142. package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
  143. package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
  144. package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
  145. package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
  146. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
  147. package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
  148. package/dist/multiplicity-DIWHvysC.d.ts +43 -0
  149. package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
  150. package/dist/multishot/index.d.ts +3 -3
  151. package/dist/multishot/index.js +1 -1
  152. package/dist/openapi.json +1 -1
  153. package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
  154. package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
  155. package/dist/package-version-D7lQHt_-.js +34 -0
  156. package/dist/package-version-D7lQHt_-.js.map +1 -0
  157. package/dist/paired-arms-D-XRF_fy.js +1045 -0
  158. package/dist/paired-arms-D-XRF_fy.js.map +1 -0
  159. package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
  160. package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
  161. package/dist/paired-tests-BHIhYVdu.js +213 -0
  162. package/dist/paired-tests-BHIhYVdu.js.map +1 -0
  163. package/dist/pareto-BqNW3LJR.d.ts +117 -0
  164. package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
  165. package/dist/pipelines/index.d.ts +3 -64
  166. package/dist/pipelines/index.d.ts.map +1 -1
  167. package/dist/pipelines/index.js +4 -284
  168. package/dist/pipelines/index.js.map +1 -1
  169. package/dist/power-and-mde-CHIrXJll.js +195 -0
  170. package/dist/power-and-mde-CHIrXJll.js.map +1 -0
  171. package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
  172. package/dist/power-preflight-DEw-uC7q.js.map +1 -0
  173. package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
  174. package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
  175. package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
  176. package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
  177. package/dist/produced-state-DU79a81m.js +586 -0
  178. package/dist/produced-state-DU79a81m.js.map +1 -0
  179. package/dist/profile-cell.d.ts +1 -1
  180. package/dist/profile-cell.js +1 -1
  181. package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
  182. package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
  183. package/dist/promotion-policy-xzA40Evo.js +186 -0
  184. package/dist/promotion-policy-xzA40Evo.js.map +1 -0
  185. package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
  186. package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
  187. package/dist/registry-oJeeI4-a.d.ts +178 -0
  188. package/dist/registry-oJeeI4-a.d.ts.map +1 -0
  189. package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
  190. package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
  191. package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
  192. package/dist/release-confidence-CxDuiAev.js.map +1 -0
  193. package/dist/reporting.d.ts +6 -5
  194. package/dist/reporting.js +7 -5
  195. package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
  196. package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
  197. package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
  198. package/dist/reward-hacking-DNgjilrV.js.map +1 -0
  199. package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
  200. package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
  201. package/dist/rl.d.ts +7 -7
  202. package/dist/rl.js +11 -10
  203. package/dist/rl.js.map +1 -1
  204. package/dist/rollout/index.d.ts +2 -2
  205. package/dist/rollout/index.js +4 -4
  206. package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
  207. package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
  208. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
  209. package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
  210. package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
  211. package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
  212. package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
  213. package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
  214. package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
  215. package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
  216. package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
  217. package/dist/run-score-lDzV0X8j.js.map +1 -0
  218. package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
  219. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
  220. package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
  221. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  222. package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
  223. package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
  224. package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
  225. package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
  226. package/dist/sequential-eprocess-CbUt2htw.js +83 -0
  227. package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
  228. package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
  229. package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
  230. package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
  231. package/dist/server-ulsOdrTI.js.map +1 -0
  232. package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
  233. package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
  234. package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
  235. package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
  236. package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
  237. package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
  238. package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
  239. package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
  240. package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
  241. package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
  242. package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
  243. package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
  244. package/dist/student-t-CvBq2mve.js +38 -0
  245. package/dist/student-t-CvBq2mve.js.map +1 -0
  246. package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
  247. package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
  248. package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
  249. package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
  250. package/dist/supervisor-run/index.d.ts +391 -3
  251. package/dist/supervisor-run/index.d.ts.map +1 -0
  252. package/dist/supervisor-run/index.js +1689 -2
  253. package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
  254. package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
  255. package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
  256. package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
  257. package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
  258. package/dist/tool-waste-BDdBZG1F.js +803 -0
  259. package/dist/tool-waste-BDdBZG1F.js.map +1 -0
  260. package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
  261. package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
  262. package/dist/trace-repair/index.d.ts +14 -5
  263. package/dist/trace-repair/index.d.ts.map +1 -1
  264. package/dist/trace-repair/index.js +35 -7
  265. package/dist/trace-repair/index.js.map +1 -1
  266. package/dist/traces.d.ts +406 -7
  267. package/dist/traces.d.ts.map +1 -0
  268. package/dist/traces.js +1011 -10
  269. package/dist/traces.js.map +1 -0
  270. package/dist/trajectory-replay/index.d.ts +16 -3
  271. package/dist/trajectory-replay/index.d.ts.map +1 -1
  272. package/dist/trajectory-replay/index.js +52 -5
  273. package/dist/trajectory-replay/index.js.map +1 -1
  274. package/dist/types-BEPZc6eo.d.ts +93 -0
  275. package/dist/types-BEPZc6eo.d.ts.map +1 -0
  276. package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
  277. package/dist/types-BI4fT3HN.js.map +1 -0
  278. package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
  279. package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
  280. package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
  281. package/dist/types-Cx3YUh2r.d.ts.map +1 -0
  282. package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
  283. package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
  284. package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
  285. package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
  286. package/dist/verdict-BndeTAh_.js +61 -0
  287. package/dist/verdict-BndeTAh_.js.map +1 -0
  288. package/dist/verdict-E4eRNf7-.d.ts +392 -0
  289. package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
  290. package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
  291. package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
  292. package/dist/wire/index.d.ts +3 -3
  293. package/dist/wire/index.d.ts.map +1 -1
  294. package/dist/wire/index.js +1 -1
  295. package/docs/charter.md +3 -3
  296. package/docs/control-runtime.md +3 -42
  297. package/docs/experiment.md +0 -1
  298. package/docs/feature-guide.md +2 -2
  299. package/docs/trace-repair-grader.md +1 -0
  300. package/docs/trajectory-replay.md +1 -0
  301. package/docs/verdicts.md +43 -0
  302. package/docs/verification-strategies.md +3 -2
  303. package/package.json +6 -11
  304. package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
  305. package/dist/analyze-runs-C30yljDJ.js.map +0 -1
  306. package/dist/baseline-CavEbRyH.d.ts +0 -136
  307. package/dist/baseline-CavEbRyH.d.ts.map +0 -1
  308. package/dist/benchmark-command-BteMFN62.js.map +0 -1
  309. package/dist/benchmarks-Dzs8CKb1.js +0 -755
  310. package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
  311. package/dist/campaign-C2TTzQII.js.map +0 -1
  312. package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
  313. package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
  314. package/dist/control.d.ts +0 -3
  315. package/dist/control.js +0 -2
  316. package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
  317. package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
  318. package/dist/default-registry-BmktKy8r.js.map +0 -1
  319. package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
  320. package/dist/experiment-tracker-CnRICnMl.js +0 -500
  321. package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
  322. package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
  323. package/dist/extract-usage-CdZdoj1s.js.map +0 -1
  324. package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
  325. package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
  326. package/dist/index-BZ3-y4YL.d.ts +0 -391
  327. package/dist/index-BZ3-y4YL.d.ts.map +0 -1
  328. package/dist/index-CQTZ-4XN.d.ts.map +0 -1
  329. package/dist/index-DPPGNJ_R.d.ts.map +0 -1
  330. package/dist/index-YE4KdKbO2.d.ts +0 -335
  331. package/dist/index-YE4KdKbO2.d.ts.map +0 -1
  332. package/dist/paired-arms-iZ08VFMN.js +0 -260
  333. package/dist/paired-arms-iZ08VFMN.js.map +0 -1
  334. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
  335. package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
  336. package/dist/prime-protocol-BfSalTfR.js.map +0 -1
  337. package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
  338. package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
  339. package/dist/promotion-policy-CrLrmys8.js.map +0 -1
  340. package/dist/proposal-findings-2GIUo1et.js.map +0 -1
  341. package/dist/propose-review-control-dSNPjFUH.js +0 -1458
  342. package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
  343. package/dist/release-report-BUYmoKo2.js.map +0 -1
  344. package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
  345. package/dist/replay-CohS93nE.js +0 -1859
  346. package/dist/replay-CohS93nE.js.map +0 -1
  347. package/dist/replay-DbhZ4Ked.d.ts +0 -834
  348. package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
  349. package/dist/reward-hacking-BDToousL.js.map +0 -1
  350. package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
  351. package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
  352. package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
  353. package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
  354. package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
  355. package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
  356. package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
  357. package/dist/server-iu0ede49.js.map +0 -1
  358. package/dist/single-run-lock-DFWHEB09.js.map +0 -1
  359. package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
  360. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
  361. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
  362. package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
  363. package/dist/statistics-ByxzSiOM.js +0 -2212
  364. package/dist/statistics-ByxzSiOM.js.map +0 -1
  365. package/dist/statistics-D6Uebe_4.d.ts +0 -968
  366. package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
  367. package/dist/supervisor-run-D_sokXcO.js +0 -1690
  368. package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
  369. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
  370. package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
  371. package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
  372. package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
  373. package/dist/tool-use-metrics-DEGMKycK.js +0 -370
  374. package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
  375. package/dist/types-D216SgwM.d.ts.map +0 -1
  376. package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
  377. package/dist/verdict-DExhxfgR.d.ts +0 -201
  378. package/dist/verdict-DExhxfgR.d.ts.map +0 -1
@@ -0,0 +1,667 @@
1
+ import { n as CaptureIntegrityError } from "./errors-Dngq5h35.js";
2
+ import { Q as applyToolSpanOtlpAttributes, i as runTraceAnalyst, tt as traceSpanKindToOpenInferenceKind } from "./kind-factory-CPmSd58s.js";
3
+ import { OPENINFERENCE_SPAN_KIND, applyLlmSpanOtlpAttributes } from "./trace-attributes.js";
4
+ import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
5
+ import { a as providerFromBaseUrl, i as defaultProviderRedactor } from "./raw-provider-sink-BQd7mzyT.js";
6
+ import { n as OtlpFileTraceStore, r as createOtlpBufferTraceStore } from "./store-otlp-CsptLYpN.js";
7
+ import { randomUUID } from "node:crypto";
8
+ import { deriveHexId, isW3CSpanId, isW3CTraceId } from "@tangle-network/agent-trace-contract";
9
+ //#region src/trace/capture-fetch.ts
10
+ /**
11
+ * Wrap a provider `fetch` and record request, response, and error events.
12
+ *
13
+ * The returned value is a plain `typeof fetch`. Capture is best-effort by
14
+ * default; set `failClosed` when telemetry loss must stop the provider call.
15
+ */
16
+ const DEFAULT_BODY_CAP = 2 * 1024 * 1024;
17
+ function headersToRecord(headers) {
18
+ if (!headers) return void 0;
19
+ const out = {};
20
+ headers.forEach((value, key) => {
21
+ out[key.toLowerCase()] = value;
22
+ });
23
+ return Object.keys(out).length > 0 ? out : void 0;
24
+ }
25
+ function parseMaybeJson(text) {
26
+ if (text.length === 0) return void 0;
27
+ try {
28
+ return JSON.parse(text);
29
+ } catch {
30
+ return text;
31
+ }
32
+ }
33
+ /** Best-effort request-body read across the `fetch` input forms. */
34
+ async function readRequestBody(input, init) {
35
+ if (typeof init?.body === "string") return parseMaybeJson(init.body);
36
+ if (init?.body != null) return void 0;
37
+ if (input instanceof Request) try {
38
+ return parseMaybeJson(await input.clone().text());
39
+ } catch {
40
+ return;
41
+ }
42
+ }
43
+ function endpointFromUrl(url, baseUrl) {
44
+ const normalisedBase = baseUrl.replace(/\/+$/, "");
45
+ if (url.startsWith(normalisedBase)) return url.slice(normalisedBase.length) || "/";
46
+ try {
47
+ return new URL(url).pathname;
48
+ } catch {
49
+ return url;
50
+ }
51
+ }
52
+ function captureFetchToRawSink(fetch, sink, ctx, opts = {}) {
53
+ const provider = ctx.provider ?? providerFromBaseUrl(ctx.baseUrl);
54
+ const redactor = opts.redactor ?? defaultProviderRedactor;
55
+ const bodyCap = opts.responseBodyByteCap ?? DEFAULT_BODY_CAP;
56
+ let warned = false;
57
+ const baseEvent = (direction, endpoint) => ({
58
+ eventId: crypto.randomUUID(),
59
+ runId: ctx.runId,
60
+ spanId: ctx.spanId,
61
+ provider,
62
+ model: ctx.model,
63
+ endpoint,
64
+ baseUrl: ctx.baseUrl,
65
+ attemptIndex: 0,
66
+ direction,
67
+ timestamp: Date.now(),
68
+ redactedFields: []
69
+ });
70
+ const record = async (event) => {
71
+ try {
72
+ await sink.record(redactor(event));
73
+ } catch (err) {
74
+ if (opts.failClosed) throw err;
75
+ if (!warned) {
76
+ warned = true;
77
+ console.warn(`captureFetchToRawSink: sink.record failed (capture is best-effort) — ${err instanceof Error ? err.message : String(err)}`);
78
+ }
79
+ }
80
+ };
81
+ return async (input, init) => {
82
+ const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
83
+ const method = (init?.method ?? (input instanceof Request ? input.method : "GET")).toUpperCase();
84
+ const endpoint = endpointFromUrl(url, ctx.baseUrl);
85
+ const reqHeaders = new Headers(init?.headers ?? (input instanceof Request ? input.headers : void 0));
86
+ await record({
87
+ ...baseEvent("request", endpoint),
88
+ requestHeaders: {
89
+ ...headersToRecord(reqHeaders),
90
+ "x-http-method": method
91
+ },
92
+ requestBody: await readRequestBody(input, init)
93
+ });
94
+ const start = Date.now();
95
+ let response;
96
+ try {
97
+ response = await fetch(input, init);
98
+ } catch (err) {
99
+ await record({
100
+ ...baseEvent("error", endpoint),
101
+ durationMs: Date.now() - start,
102
+ errorMessage: err instanceof Error ? err.message : String(err)
103
+ });
104
+ throw err;
105
+ }
106
+ let responseBody;
107
+ let rawText;
108
+ const redactedFields = [];
109
+ try {
110
+ rawText = await response.clone().text();
111
+ if (rawText.length > bodyCap) {
112
+ responseBody = rawText.slice(0, bodyCap);
113
+ redactedFields.push("body_truncated");
114
+ } else responseBody = parseMaybeJson(rawText);
115
+ } catch {
116
+ responseBody = void 0;
117
+ }
118
+ if (opts.onUsage && rawText !== void 0) try {
119
+ const parsedForUsage = parseMaybeJson(rawText);
120
+ const usage = extractUsage(parsedForUsage) ?? (typeof parsedForUsage === "string" ? extractUsageFromSse(rawText, { mode: opts.sseUsageMode }) : null);
121
+ if (usage) opts.onUsage(usage, ctx);
122
+ } catch (err) {
123
+ if (opts.failClosed) throw err;
124
+ }
125
+ await record({
126
+ ...baseEvent("response", endpoint),
127
+ durationMs: Date.now() - start,
128
+ statusCode: response.status,
129
+ responseHeaders: headersToRecord(response.headers),
130
+ responseBody,
131
+ redactedFields
132
+ });
133
+ return response;
134
+ };
135
+ }
136
+ //#endregion
137
+ //#region src/trace/wire-ids.ts
138
+ /**
139
+ * The ONE mapping from agent-eval's human-readable ids (run ids, span labels)
140
+ * to W3C/OTLP wire ids. Every exporter in this package MUST route through
141
+ * these two functions — two exporters with private paddings once produced
142
+ * DIFFERENT trace ids for the same run, and one emitted invalid hex embedding
143
+ * the raw run id in the wire id (tangle-network/agent-runtime#694).
144
+ *
145
+ * Semantics:
146
+ * - an id that is ALREADY a valid W3C id passes through unchanged, so a
147
+ * trace id received from an inbound `traceparent` survives the round-trip
148
+ * and cross-process correlation is preserved;
149
+ * - anything else is derived with the contract's `deriveHexId`, the only
150
+ * legal derivation — deterministic, so every process that derives from the
151
+ * same human id mints the SAME wire id.
152
+ */
153
+ /** 32-hex W3C trace id for any id string. */
154
+ function traceIdForWire(id) {
155
+ return isW3CTraceId(id) ? id : deriveHexId(id, 16);
156
+ }
157
+ /** 16-hex W3C span id for any id string. */
158
+ function spanIdForWire(id) {
159
+ return isW3CSpanId(id) ? id : deriveHexId(id, 8);
160
+ }
161
+ //#endregion
162
+ //#region src/trace/otel.ts
163
+ /**
164
+ * OpenTelemetry JSON export — maps TraceSchema v1 to OTLP/JSON so
165
+ * traces render natively in Jaeger / Honeycomb / Langfuse / Grafana.
166
+ *
167
+ * Wire format only. We do NOT depend on the @opentelemetry SDK — that
168
+ * would drag in polyfills incompatible with Workers/Edge. Consumers
169
+ * push the JSON to their collector of choice via HTTP.
170
+ *
171
+ * Reference: OTLP 1.3.2 (ResourceSpans / ScopeSpans / Span).
172
+ */
173
+ const OTEL_AGENT_EVAL_SCOPE = {
174
+ name: "@tangle-network/agent-eval",
175
+ version: "0.3.0"
176
+ };
177
+ /** Export a single run's spans + events in OTLP/JSON. */
178
+ async function exportRunAsOtlp(store, runId, resourceAttrs = {}) {
179
+ const run = await store.getRun(runId);
180
+ if (!run) throw new Error(`run ${runId} not found`);
181
+ const spans = await store.spans({ runId });
182
+ const events = await store.events({ runId });
183
+ const eventsBySpan = /* @__PURE__ */ new Map();
184
+ for (const e of events) {
185
+ if (!e.spanId) continue;
186
+ const arr = eventsBySpan.get(e.spanId) ?? [];
187
+ arr.push(e);
188
+ eventsBySpan.set(e.spanId, arr);
189
+ }
190
+ const traceId = runToTraceId(run);
191
+ const otlpSpans = spans.map((s) => spanToOtlp(s, traceId, eventsBySpan.get(s.spanId) ?? []));
192
+ return { resourceSpans: [{
193
+ resource: { attributes: toAttributes({
194
+ "service.name": "agent-eval",
195
+ "run.id": run.runId,
196
+ "run.scenario_id": run.scenarioId,
197
+ "run.variant_id": run.variantId ?? "",
198
+ "run.dataset_version": run.datasetVersion ?? "",
199
+ "run.code_sha": run.codeSha ?? "",
200
+ "run.model_fingerprint": run.modelFingerprint ?? "",
201
+ ...resourceAttrs
202
+ }) },
203
+ scopeSpans: [{
204
+ scope: OTEL_AGENT_EVAL_SCOPE,
205
+ spans: otlpSpans
206
+ }]
207
+ }] };
208
+ }
209
+ function spanToOtlp(span, traceId, events) {
210
+ const endedAt = span.endedAt ?? span.startedAt;
211
+ return {
212
+ traceId,
213
+ spanId: spanIdForWire(span.spanId),
214
+ parentSpanId: span.parentSpanId ? spanIdForWire(span.parentSpanId) : void 0,
215
+ name: span.name,
216
+ kind: 1,
217
+ startTimeUnixNano: msToNs(span.startedAt),
218
+ endTimeUnixNano: msToNs(endedAt),
219
+ attributes: toAttributes(flattenSpanAttributes(span)),
220
+ events: events.map((e) => ({
221
+ timeUnixNano: msToNs(e.timestamp),
222
+ name: e.kind,
223
+ attributes: toAttributes(flattenPayload(e.payload))
224
+ })),
225
+ status: span.status === "error" ? {
226
+ code: 2,
227
+ message: span.error
228
+ } : { code: 1 }
229
+ };
230
+ }
231
+ function flattenSpanAttributes(span) {
232
+ const base = {};
233
+ if (span.attributes) {
234
+ for (const [k, v] of Object.entries(span.attributes)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") base[k] = v;
235
+ }
236
+ base[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
237
+ if (span.kind === "llm") applyLlmSpanOtlpAttributes(base, span);
238
+ else if (span.kind === "tool") applyToolSpanOtlpAttributes(base, span);
239
+ else if (span.kind === "retrieval") {
240
+ base["retrieval.query"] = span.query;
241
+ base["retrieval.hits"] = span.hits.length;
242
+ } else if (span.kind === "judge") {
243
+ base["judge.id"] = span.judgeId;
244
+ base["judge.dimension"] = span.dimension;
245
+ base["judge.score"] = span.score;
246
+ base["judge.target_span_id"] = span.targetSpanId;
247
+ } else if (span.kind === "sandbox") {
248
+ if (span.image) base["sandbox.image"] = span.image;
249
+ if (span.exitCode !== void 0) base["sandbox.exit_code"] = span.exitCode;
250
+ if (span.testsPassed !== void 0) base["sandbox.tests_passed"] = span.testsPassed;
251
+ if (span.testsTotal !== void 0) base["sandbox.tests_total"] = span.testsTotal;
252
+ }
253
+ return base;
254
+ }
255
+ function flattenPayload(payload) {
256
+ const out = {};
257
+ for (const [k, v] of Object.entries(payload)) if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") out[k] = v;
258
+ else out[k] = JSON.stringify(v);
259
+ return out;
260
+ }
261
+ function toAttributes(record) {
262
+ return Object.entries(record).map(([key, value]) => ({
263
+ key,
264
+ value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
265
+ }));
266
+ }
267
+ function msToNs(ms) {
268
+ return (BigInt(Math.floor(ms)) * 1000000n).toString();
269
+ }
270
+ function runToTraceId(run) {
271
+ return traceIdForWire(run.runId);
272
+ }
273
+ //#endregion
274
+ //#region src/trace-analyst/prompts.ts
275
+ /** General policy for recursive, evidence-backed trace analysis. */
276
+ const TRACE_ANALYST_ACTOR_DESCRIPTION = `Answer the question by inspecting the OTLP trace dataset with the available tools.
277
+
278
+ 1. Call getDatasetOverview first. Use its real trace ids and dataset size to plan the investigation.
279
+ 2. Narrow with queryTraces and countTraces before scanning large payloads.
280
+ 3. For a small trace, use viewTrace. For a large trace, use searchTrace and then viewSpans or searchSpan.
281
+ 4. Never invent a trace id, span id, tool result, error, frequency, or final outcome.
282
+ 5. When a search reports has_more, refine the query before drawing a conclusion.
283
+ 6. Use llm_query only over evidence already loaded. A recursive query cannot inspect traces itself.
284
+ 7. Cite exact evidence URIs returned by the tools. Include a short exact excerpt when it supports the claim.
285
+ 8. Return no finding when the available evidence cannot support one.
286
+
287
+ The prose answer must directly answer the question and state important uncertainty.
288
+ The findings array contains only actionable or decision-relevant claims supported by inspected evidence.`;
289
+ const TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION = "trace-analyst-research-v1-2026-07-30";
290
+ //#endregion
291
+ //#region src/trace-analyst/analyst.ts
292
+ /**
293
+ * Answer one question by recursively inspecting a trace store.
294
+ *
295
+ * The returned answer, cited findings, engine steps, call counts, and runtime
296
+ * identity are one audit record. A direct one-shot model call is not used.
297
+ */
298
+ async function analyzeTraces(input, options) {
299
+ if (typeof input.question !== "string" || !input.question.trim()) throw new TypeError("analyzeTraces: input.question must be a non-empty string");
300
+ const id = input.id?.trim() || "trace-analysis";
301
+ const store = typeof options.source === "string" ? new OtlpFileTraceStore({ path: options.source }) : options.source;
302
+ if (store instanceof OtlpFileTraceStore) await store.ensureIndexed(options.signal ? { signal: options.signal } : void 0);
303
+ return runTraceAnalyst({
304
+ definition: {
305
+ id,
306
+ description: input.description?.trim() || "Answers a caller-defined question by recursively inspecting trace evidence.",
307
+ area: input.area?.trim() || "trace-analysis",
308
+ version: TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
309
+ question: input.question,
310
+ instructions: options.instructions ?? TRACE_ANALYST_ACTOR_DESCRIPTION,
311
+ toolGroup: options.toolGroup ?? "all",
312
+ limits: options.limits
313
+ },
314
+ engine: options.engine,
315
+ store,
316
+ context: {
317
+ runId: options.runId ?? id,
318
+ correlationId: randomUUID(),
319
+ budgetUsd: options.budgetUsd,
320
+ costLedger: options.costLedger,
321
+ costPhase: options.costPhase ?? "trace-analysis",
322
+ priorFindings: options.priorFindings,
323
+ upstreamFindings: options.upstreamFindings,
324
+ recordUsage: options.recordUsage,
325
+ tags: options.tags,
326
+ log: options.log,
327
+ signal: options.signal
328
+ }
329
+ });
330
+ }
331
+ //#endregion
332
+ //#region src/trace-analyst/insights.ts
333
+ const DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
334
+ "and",
335
+ "advanced",
336
+ "app",
337
+ "build",
338
+ "create",
339
+ "easy",
340
+ "expert",
341
+ "extreme",
342
+ "for",
343
+ "from",
344
+ "hard",
345
+ "implementation",
346
+ "integrate",
347
+ "medium",
348
+ "project",
349
+ "task",
350
+ "the",
351
+ "this",
352
+ "with",
353
+ "workflow"
354
+ ]);
355
+ function tokenizeDomainWords(value) {
356
+ return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
357
+ }
358
+ function inferDomainKeywords(suite) {
359
+ const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
360
+ const source = [
361
+ suite.name,
362
+ suite.collectionId ?? "",
363
+ ...suite.tasks.flatMap((task) => [
364
+ task.id,
365
+ task.name,
366
+ task.prompt ?? "",
367
+ task.difficulty ?? "",
368
+ ...task.tags ?? [],
369
+ ...task.gaps ?? []
370
+ ])
371
+ ].join(" ");
372
+ const counts = /* @__PURE__ */ new Map();
373
+ for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
374
+ return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
375
+ }
376
+ function domainEvidencePattern(keywords) {
377
+ const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
378
+ return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
379
+ }
380
+ function describeTraceInsightScope(suite) {
381
+ const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
382
+ const tags = /* @__PURE__ */ new Map();
383
+ for (const task of suite.tasks) for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
384
+ const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
385
+ if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
386
+ return `${taskLabel} across ${[...new Set(suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value)))].join(", ") || "the selected benchmark scope"}.`;
387
+ }
388
+ function planTraceInsightQuestions(input) {
389
+ const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
390
+ const hasMultipleShots = input.suite.tasks.some((task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap)));
391
+ const questions = [
392
+ {
393
+ id: "execution-path",
394
+ question: "What did the worker actually do before the first meaningful implementation edit?",
395
+ why: "Separates grounded execution from polished but shallow output."
396
+ },
397
+ {
398
+ id: "research-grounding",
399
+ question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
400
+ why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
401
+ },
402
+ {
403
+ id: "domain-proof",
404
+ question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
405
+ why: "Keeps product-quality claims tied to concrete evidence."
406
+ },
407
+ {
408
+ id: "root-cause",
409
+ question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
410
+ why: "Turns trace observations into actionable ownership."
411
+ },
412
+ {
413
+ id: "evidence-quality",
414
+ question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
415
+ why: "Prevents unsupported customer-report conclusions."
416
+ }
417
+ ];
418
+ if (hasMultipleShots) questions.push({
419
+ id: "reviewer-lift",
420
+ question: "Where did reviewer feedback improve score, stall, or regress across shots?",
421
+ why: "Shows whether the driver loop is learning or merely repeating work."
422
+ });
423
+ if (hasFailures) questions.push({
424
+ id: "optimization-targets",
425
+ question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
426
+ why: "Connects benchmark evidence to the optimization loop."
427
+ });
428
+ return questions;
429
+ }
430
+ function buildTraceInsightContext(input) {
431
+ return {
432
+ suite: input.suite,
433
+ scope: describeTraceInsightScope(input.suite),
434
+ keywords: inferDomainKeywords(input.suite),
435
+ questions: planTraceInsightQuestions(input),
436
+ panel: defaultTraceInsightPanel(),
437
+ findings: input.findings ?? [],
438
+ agent: input.agent ?? null,
439
+ totals: input.totals ?? null
440
+ };
441
+ }
442
+ function scoreTraceInsightReadiness(context) {
443
+ const failedTasks = context.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied");
444
+ const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
445
+ const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
446
+ const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
447
+ const gates = [
448
+ {
449
+ id: "domain-context",
450
+ label: "Domain context inferred",
451
+ passed: context.keywords.length > 0,
452
+ severity: "high",
453
+ detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
454
+ },
455
+ {
456
+ id: "panel-coverage",
457
+ label: "Analyst panel planned",
458
+ passed: context.panel.length >= 4 && context.questions.length >= 5,
459
+ severity: "high",
460
+ detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
461
+ },
462
+ {
463
+ id: "failure-coverage",
464
+ label: "Failures mapped to findings",
465
+ passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= .5,
466
+ severity: "critical",
467
+ detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
468
+ },
469
+ {
470
+ id: "gap-evidence",
471
+ label: "Task gaps captured",
472
+ passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= .5,
473
+ severity: "medium",
474
+ detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
475
+ }
476
+ ];
477
+ const penalty = gates.reduce((sum, gate) => {
478
+ if (gate.passed) return sum;
479
+ if (gate.severity === "critical") return sum + 35;
480
+ if (gate.severity === "high") return sum + 20;
481
+ if (gate.severity === "medium") return sum + 10;
482
+ return sum + 5;
483
+ }, 0);
484
+ const score = Math.max(0, Math.min(1, 1 - penalty / 100));
485
+ return {
486
+ score,
487
+ grade: score >= .9 ? "external-ready" : score >= .7 ? "internal-review" : "raw-analysis",
488
+ gates
489
+ };
490
+ }
491
+ function defaultTraceInsightPanel() {
492
+ return [
493
+ {
494
+ id: "trace-forensics",
495
+ name: "Trace Forensics",
496
+ responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
497
+ },
498
+ {
499
+ id: "root-cause",
500
+ name: "Root Cause",
501
+ responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
502
+ },
503
+ {
504
+ id: "optimization",
505
+ name: "Optimization",
506
+ responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
507
+ },
508
+ {
509
+ id: "external-evidence",
510
+ name: "External Evidence",
511
+ responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
512
+ }
513
+ ];
514
+ }
515
+ function buildTraceInsightPrompt(input) {
516
+ const context = buildTraceInsightContext(input);
517
+ const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
518
+ return `Analyze this benchmark run and produce evidence-backed trace intelligence.
519
+
520
+ Audience:
521
+ - internal AI/product leadership
522
+ - possible customer-facing report for ${input.suite.name}
523
+
524
+ Investigation plan:
525
+ ${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
526
+
527
+ Analyst panel:
528
+ ${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
529
+
530
+ If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
531
+
532
+ Required output:
533
+ 1. Executive verdict: what this run proves and does not prove.
534
+ 2. The investigation questions you answered and the evidence used.
535
+ 3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
536
+ 4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
537
+ 5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
538
+ 6. What is safe for an external report versus what must stay internal.
539
+ 7. One rerun plan that would validate lift after optimization.
540
+
541
+ Budget:
542
+ - Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
543
+ - Prefer traces named in the failure summary over broad exploration.
544
+ - Do not do exhaustive trace sweeps.
545
+ - Return the final report as soon as the taxonomy and examples are supported.
546
+
547
+ Run summary:
548
+ ${JSON.stringify({
549
+ suite: input.suite.name,
550
+ scope: context.scope,
551
+ inferredKeywords: context.keywords,
552
+ agent: context.agent,
553
+ totals: context.totals,
554
+ findings: context.findings.map((finding) => ({
555
+ kind: finding.kind,
556
+ severity: finding.severity,
557
+ taskCount: finding.taskIds.length,
558
+ proposedFixClass: finding.proposedFixClass
559
+ })),
560
+ failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
561
+ task: task.id,
562
+ difficulty: task.difficulty,
563
+ outcome: task.outcome,
564
+ score: task.score,
565
+ gaps: task.gaps ?? []
566
+ }))
567
+ }, null, 2)}
568
+
569
+ Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
570
+ }
571
+ //#endregion
572
+ //#region src/trace/otlp-flat.ts
573
+ function createOtlpFlatLine(input) {
574
+ return {
575
+ trace_id: input.traceId,
576
+ span_id: input.spanId,
577
+ parent_span_id: input.parentSpanId,
578
+ name: input.name,
579
+ kind: input.kind,
580
+ start_time: input.startTime,
581
+ end_time: input.endTime,
582
+ status: {
583
+ code: input.statusCode,
584
+ ...input.statusMessage !== void 0 ? { message: input.statusMessage } : {}
585
+ },
586
+ resource: input.resource,
587
+ attributes: input.attributes,
588
+ ...input.events && input.events.length > 0 ? { events: input.events } : {}
589
+ };
590
+ }
591
+ /** Map the canonical trace status while letting each caller choose its legacy default. */
592
+ function spanStatusToOtlp(status, error, defaultCode) {
593
+ if (status === "error" || error) return "STATUS_CODE_ERROR";
594
+ if (status === "ok") return "STATUS_CODE_OK";
595
+ return defaultCode;
596
+ }
597
+ /** Convert epoch milliseconds, returning `undefined` for invalid or out-of-range values. */
598
+ function epochMillisToIso(value) {
599
+ if (!Number.isFinite(value)) return void 0;
600
+ try {
601
+ return new Date(value).toISOString();
602
+ } catch {
603
+ return;
604
+ }
605
+ }
606
+ //#endregion
607
+ //#region src/trace-analyst/store-tool-spans.ts
608
+ /** Missing tool spans cannot distinguish a tool-free run from broken capture. */
609
+ var ToolTraceMissingError = class extends CaptureIntegrityError {
610
+ constructor() {
611
+ super("toolSpansToTraceAnalysisStore: no tool spans supplied; trace evidence is missing");
612
+ }
613
+ };
614
+ /**
615
+ * Snapshot canonical tool spans into the read interface used by trace analysts.
616
+ * One run becomes one trace while arguments, results, errors, and timing remain searchable.
617
+ */
618
+ function toolSpansToTraceAnalysisStore(spans, options = {}) {
619
+ if (!spans || spans.length === 0) throw new ToolTraceMissingError();
620
+ const seen = /* @__PURE__ */ new Set();
621
+ const lines = spans.map((span, index) => {
622
+ assertToolSpanIdentity(span, index);
623
+ const identity = `${span.runId}\u0000${span.spanId}`;
624
+ if (seen.has(identity)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: duplicate span '${span.spanId}' in run '${span.runId}'`);
625
+ seen.add(identity);
626
+ const attributes = { ...span.attributes ?? {} };
627
+ applyToolSpanOtlpAttributes(attributes, span);
628
+ attributes[OPENINFERENCE_SPAN_KIND] = "TOOL";
629
+ const endedAt = span.endedAt ?? span.startedAt + (span.latencyMs ?? 0);
630
+ const line = createOtlpFlatLine({
631
+ traceId: span.runId,
632
+ spanId: span.spanId,
633
+ parentSpanId: span.parentSpanId ?? null,
634
+ name: span.name,
635
+ kind: "SPAN_KIND_INTERNAL",
636
+ startTime: toolSpanTimeIso(span.startedAt, span.spanId, "startedAt"),
637
+ endTime: toolSpanTimeIso(endedAt, span.spanId, "endedAt"),
638
+ statusCode: spanStatusToOtlp(span.status, span.error, "STATUS_CODE_UNSET"),
639
+ statusMessage: span.error,
640
+ resource: { attributes: {} },
641
+ attributes
642
+ });
643
+ try {
644
+ return JSON.stringify(line);
645
+ } catch (cause) {
646
+ throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' in run '${span.runId}' is not JSON-serializable`, { cause });
647
+ }
648
+ });
649
+ return createOtlpBufferTraceStore(Buffer.from(`${lines.join("\n")}\n`, "utf8"), options);
650
+ }
651
+ function assertToolSpanIdentity(span, index) {
652
+ if (span.kind !== "tool") throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span at index ${index} has kind '${String(span.kind)}', not 'tool'`);
653
+ if (!span.runId || !span.spanId || !span.name || !span.toolName) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span at index ${index} is missing runId, spanId, name, or toolName`);
654
+ if (!Number.isFinite(span.startedAt)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid startedAt`);
655
+ if (span.endedAt !== void 0 && !Number.isFinite(span.endedAt)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid endedAt`);
656
+ if (span.endedAt !== void 0 && span.endedAt < span.startedAt) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' ends before it starts`);
657
+ if (span.latencyMs !== void 0 && (!Number.isFinite(span.latencyMs) || span.latencyMs < 0)) throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${span.spanId}' has invalid latencyMs`);
658
+ }
659
+ function toolSpanTimeIso(value, spanId, field) {
660
+ const iso = epochMillisToIso(value);
661
+ if (iso) return iso;
662
+ throw new CaptureIntegrityError(`toolSpansToTraceAnalysisStore: span '${spanId}' has invalid ${field}`);
663
+ }
664
+ //#endregion
665
+ export { captureFetchToRawSink as S, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION as _, spanStatusToOtlp as a, spanIdForWire as b, defaultTraceInsightPanel as c, inferDomainKeywords as d, planTraceInsightQuestions as f, TRACE_ANALYST_ACTOR_DESCRIPTION as g, analyzeTraces as h, epochMillisToIso as i, describeTraceInsightScope as l, tokenizeDomainWords as m, toolSpansToTraceAnalysisStore as n, buildTraceInsightContext as o, scoreTraceInsightReadiness as p, createOtlpFlatLine as r, buildTraceInsightPrompt as s, ToolTraceMissingError as t, domainEvidencePattern as u, OTEL_AGENT_EVAL_SCOPE as v, traceIdForWire as x, exportRunAsOtlp as y };
666
+
667
+ //# sourceMappingURL=store-tool-spans-Cq9mFd-q.js.map