@opensearch-project/agent-health 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (551) hide show
  1. package/README.md +36 -5
  2. package/cli/dist/index.js +8292 -2938
  3. package/deployment/cloudformation/agent-health-observability.yaml +223 -32
  4. package/dist/assets/index-BfxtxmKc.css +1 -0
  5. package/dist/assets/index-CrjAfDHu.js +243 -0
  6. package/dist/index.html +2 -2
  7. package/docs/ARCHITECTURE.md +450 -0
  8. package/docs/BACKEND_JOB_QUEUE.md +405 -0
  9. package/docs/CLAUDE_CODE_TELEMETRY.md +283 -0
  10. package/docs/CLI.md +474 -0
  11. package/docs/CODING_AGENT_ANALYTICS.md +298 -0
  12. package/docs/CONFIGURATION.md +388 -0
  13. package/docs/CONNECTORS.md +536 -0
  14. package/docs/INSTRUMENT_WITH_OTEL.md +397 -0
  15. package/docs/ML-COMMONS-SETUP.md +289 -0
  16. package/docs/NPX_PACKAGING.md +195 -0
  17. package/docs/PERFORMANCE-MONITORING.md +200 -0
  18. package/docs/PERFORMANCE.md +390 -0
  19. package/docs/PI_PROFILING.md +169 -0
  20. package/docs/PLAN-non-agui-agent-support.md +525 -0
  21. package/docs/SDK.md +626 -0
  22. package/docs/SKILLS.md +264 -0
  23. package/docs/STORAGE_INDEX_FIELD_LIMITS.md +216 -0
  24. package/docs/blogs/2026-02-28-opensearch-agent-health.md +200 -0
  25. package/docs/blogs/getting-started-blog.md +608 -0
  26. package/docs/diagrams/Agent-health.excalidraw +5656 -0
  27. package/docs/diagrams/architecture.png +0 -0
  28. package/docs/plans/field-redesign.md +468 -0
  29. package/docs/rfcs/001-coding-agent-analytics.md +374 -0
  30. package/docs/rfcs/002-enterprise-leaderboard.md +267 -0
  31. package/docs/rfcs/003-remote-aggregation.md +146 -0
  32. package/docs/rfcs/004-test-sdk-v2.md +599 -0
  33. package/docs/skills/AGENT_HEALTH.md +652 -0
  34. package/docs/skills/AGENT_PROFILE.md +191 -0
  35. package/docs/skills/add-connector/SKILL.md +68 -0
  36. package/docs/skills/agent-health-profile/SKILL.md +40 -0
  37. package/docs/skills/config-auth/SKILL.md +194 -0
  38. package/docs/skills/config-auth/evals/evals.json +35 -0
  39. package/docs/skills/create-pr/SKILL.md +73 -0
  40. package/docs/skills/instrument-otel/SKILL.md +84 -0
  41. package/docs/skills/write-test/SKILL.md +124 -0
  42. package/docs/ui prd.md +376 -0
  43. package/examples/README.md +53 -0
  44. package/examples/config/agent-health.config.example.ts +155 -0
  45. package/examples/connectors/echo-connector.ts +131 -0
  46. package/examples/eval-files/demo.eval.js +128 -0
  47. package/examples/eval-files/ops-rca-classification.eval.js +71 -0
  48. package/examples/eval-files/ops-rca-evaluator.json +15 -0
  49. package/examples/eval-files/sdk-demo.eval.js +72 -0
  50. package/examples/eval-files/sdk-describe-demo.eval.js +50 -0
  51. package/examples/eval-files/sdk-hooks-demo.eval.js +99 -0
  52. package/examples/pi-profiling/README.md +77 -0
  53. package/examples/pi-profiling/agent-health-profile.ts +417 -0
  54. package/lib/dist/lib/agentUtils.d.ts +29 -0
  55. package/lib/dist/lib/agentUtils.d.ts.map +1 -0
  56. package/lib/dist/lib/agentUtils.js +43 -0
  57. package/lib/dist/lib/agentUtils.js.map +1 -0
  58. package/lib/dist/lib/bedrockCompat.d.ts +27 -0
  59. package/lib/dist/lib/bedrockCompat.d.ts.map +1 -0
  60. package/lib/dist/lib/bedrockCompat.js +83 -0
  61. package/lib/dist/lib/bedrockCompat.js.map +1 -0
  62. package/lib/dist/lib/benchmarkExport.d.ts +14 -0
  63. package/lib/dist/lib/benchmarkExport.d.ts.map +1 -0
  64. package/lib/dist/lib/benchmarkExport.js +41 -0
  65. package/lib/dist/lib/benchmarkExport.js.map +1 -0
  66. package/lib/dist/lib/benchmarkImage.d.ts +52 -0
  67. package/lib/dist/lib/benchmarkImage.d.ts.map +1 -0
  68. package/lib/dist/lib/benchmarkImage.js +113 -0
  69. package/lib/dist/lib/benchmarkImage.js.map +1 -0
  70. package/lib/dist/lib/benchmarkVersionUtils.d.ts +50 -0
  71. package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -0
  72. package/lib/dist/lib/benchmarkVersionUtils.js +89 -0
  73. package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -0
  74. package/lib/dist/lib/chunkedFetch.d.ts +18 -0
  75. package/lib/dist/lib/chunkedFetch.d.ts.map +1 -0
  76. package/lib/dist/lib/chunkedFetch.js +40 -0
  77. package/lib/dist/lib/chunkedFetch.js.map +1 -0
  78. package/lib/dist/lib/comparisonInsights.d.ts +104 -0
  79. package/lib/dist/lib/comparisonInsights.d.ts.map +1 -0
  80. package/lib/dist/lib/comparisonInsights.js +212 -0
  81. package/lib/dist/lib/comparisonInsights.js.map +1 -0
  82. package/lib/dist/lib/config/defineConfig.d.ts +27 -0
  83. package/lib/dist/lib/config/defineConfig.d.ts.map +1 -0
  84. package/lib/dist/lib/config/defineConfig.js +28 -0
  85. package/lib/dist/lib/config/defineConfig.js.map +1 -0
  86. package/lib/dist/lib/config/index.d.ts +9 -0
  87. package/lib/dist/lib/config/index.d.ts.map +1 -0
  88. package/lib/dist/lib/config/index.js +8 -0
  89. package/lib/dist/lib/config/index.js.map +1 -0
  90. package/lib/dist/lib/config/loader.d.ts +39 -0
  91. package/lib/dist/lib/config/loader.d.ts.map +1 -0
  92. package/lib/dist/lib/config/loader.js +263 -0
  93. package/lib/dist/lib/config/loader.js.map +1 -0
  94. package/lib/dist/lib/config/statePaths.d.ts +61 -0
  95. package/lib/dist/lib/config/statePaths.d.ts.map +1 -0
  96. package/lib/dist/lib/config/statePaths.js +188 -0
  97. package/lib/dist/lib/config/statePaths.js.map +1 -0
  98. package/lib/dist/lib/config/types.d.ts +245 -0
  99. package/lib/dist/lib/config/types.d.ts.map +1 -0
  100. package/lib/dist/lib/config/types.js +6 -0
  101. package/lib/dist/lib/config/types.js.map +1 -0
  102. package/lib/dist/lib/config.d.ts +39 -0
  103. package/lib/dist/lib/config.d.ts.map +1 -0
  104. package/lib/dist/lib/config.js +118 -0
  105. package/lib/dist/lib/config.js.map +1 -0
  106. package/lib/dist/lib/constants.d.ts +81 -0
  107. package/lib/dist/lib/constants.d.ts.map +1 -0
  108. package/lib/dist/lib/constants.js +374 -0
  109. package/lib/dist/lib/constants.js.map +1 -0
  110. package/lib/dist/lib/contextFormat.d.ts +26 -0
  111. package/lib/dist/lib/contextFormat.d.ts.map +1 -0
  112. package/lib/dist/lib/contextFormat.js +28 -0
  113. package/lib/dist/lib/contextFormat.js.map +1 -0
  114. package/lib/dist/lib/contextUtilization.d.ts +23 -0
  115. package/lib/dist/lib/contextUtilization.d.ts.map +1 -0
  116. package/lib/dist/lib/contextUtilization.js +72 -0
  117. package/lib/dist/lib/contextUtilization.js.map +1 -0
  118. package/lib/dist/lib/dashboardMetrics.d.ts +87 -0
  119. package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -0
  120. package/lib/dist/lib/dashboardMetrics.js +242 -0
  121. package/lib/dist/lib/dashboardMetrics.js.map +1 -0
  122. package/lib/dist/lib/dataSourceConfig.d.ts +108 -0
  123. package/lib/dist/lib/dataSourceConfig.d.ts.map +1 -0
  124. package/lib/dist/lib/dataSourceConfig.js +166 -0
  125. package/lib/dist/lib/dataSourceConfig.js.map +1 -0
  126. package/lib/dist/lib/debug.d.ts +26 -0
  127. package/lib/dist/lib/debug.d.ts.map +1 -0
  128. package/lib/dist/lib/debug.js +132 -0
  129. package/lib/dist/lib/debug.js.map +1 -0
  130. package/lib/dist/lib/diagnostics.d.ts +28 -0
  131. package/lib/dist/lib/diagnostics.d.ts.map +1 -0
  132. package/lib/dist/lib/diagnostics.js +65 -0
  133. package/lib/dist/lib/diagnostics.js.map +1 -0
  134. package/lib/dist/lib/envCompat.d.ts +27 -0
  135. package/lib/dist/lib/envCompat.d.ts.map +1 -0
  136. package/lib/dist/lib/envCompat.js +82 -0
  137. package/lib/dist/lib/envCompat.js.map +1 -0
  138. package/lib/dist/lib/evaluationRerun.d.ts +63 -0
  139. package/lib/dist/lib/evaluationRerun.d.ts.map +1 -0
  140. package/lib/dist/lib/evaluationRerun.js +85 -0
  141. package/lib/dist/lib/evaluationRerun.js.map +1 -0
  142. package/lib/dist/lib/findPackageRoot.d.ts +7 -0
  143. package/lib/dist/lib/findPackageRoot.d.ts.map +1 -0
  144. package/lib/dist/lib/findPackageRoot.js +57 -0
  145. package/lib/dist/lib/findPackageRoot.js.map +1 -0
  146. package/lib/dist/lib/hooks.d.ts +36 -0
  147. package/lib/dist/lib/hooks.d.ts.map +1 -0
  148. package/lib/dist/lib/hooks.js +112 -0
  149. package/lib/dist/lib/hooks.js.map +1 -0
  150. package/lib/dist/lib/index.d.ts +47 -0
  151. package/lib/dist/lib/index.d.ts.map +1 -0
  152. package/lib/dist/lib/index.js +62 -0
  153. package/lib/dist/lib/index.js.map +1 -0
  154. package/lib/dist/lib/labels.d.ts +90 -0
  155. package/lib/dist/lib/labels.d.ts.map +1 -0
  156. package/lib/dist/lib/labels.js +158 -0
  157. package/lib/dist/lib/labels.js.map +1 -0
  158. package/lib/dist/lib/markdown.d.ts +16 -0
  159. package/lib/dist/lib/markdown.d.ts.map +1 -0
  160. package/lib/dist/lib/markdown.js +42 -0
  161. package/lib/dist/lib/markdown.js.map +1 -0
  162. package/lib/dist/lib/matchers/expect.d.ts +3 -0
  163. package/lib/dist/lib/matchers/expect.d.ts.map +1 -0
  164. package/lib/dist/lib/matchers/expect.js +225 -0
  165. package/lib/dist/lib/matchers/expect.js.map +1 -0
  166. package/lib/dist/lib/matchers/index.d.ts +8 -0
  167. package/lib/dist/lib/matchers/index.d.ts.map +1 -0
  168. package/lib/dist/lib/matchers/index.js +9 -0
  169. package/lib/dist/lib/matchers/index.js.map +1 -0
  170. package/lib/dist/lib/matchers/judgeAccessor.d.ts +113 -0
  171. package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -0
  172. package/lib/dist/lib/matchers/judgeAccessor.js +183 -0
  173. package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -0
  174. package/lib/dist/lib/matchers/session.d.ts +39 -0
  175. package/lib/dist/lib/matchers/session.d.ts.map +1 -0
  176. package/lib/dist/lib/matchers/session.js +116 -0
  177. package/lib/dist/lib/matchers/session.js.map +1 -0
  178. package/lib/dist/lib/matchers/traces.d.ts +70 -0
  179. package/lib/dist/lib/matchers/traces.d.ts.map +1 -0
  180. package/lib/dist/lib/matchers/traces.js +236 -0
  181. package/lib/dist/lib/matchers/traces.js.map +1 -0
  182. package/lib/dist/lib/matchers/tracesPricing.d.ts +38 -0
  183. package/lib/dist/lib/matchers/tracesPricing.d.ts.map +1 -0
  184. package/lib/dist/lib/matchers/tracesPricing.js +64 -0
  185. package/lib/dist/lib/matchers/tracesPricing.js.map +1 -0
  186. package/lib/dist/lib/matchers/types.d.ts +75 -0
  187. package/lib/dist/lib/matchers/types.d.ts.map +1 -0
  188. package/lib/dist/lib/matchers/types.js +6 -0
  189. package/lib/dist/lib/matchers/types.js.map +1 -0
  190. package/lib/dist/lib/packagePaths.d.ts +29 -0
  191. package/lib/dist/lib/packagePaths.d.ts.map +1 -0
  192. package/lib/dist/lib/packagePaths.js +63 -0
  193. package/lib/dist/lib/packagePaths.js.map +1 -0
  194. package/lib/dist/lib/performance.d.ts +51 -0
  195. package/lib/dist/lib/performance.d.ts.map +1 -0
  196. package/lib/dist/lib/performance.js +159 -0
  197. package/lib/dist/lib/performance.js.map +1 -0
  198. package/lib/dist/lib/portConfig.d.ts +29 -0
  199. package/lib/dist/lib/portConfig.d.ts.map +1 -0
  200. package/lib/dist/lib/portConfig.js +64 -0
  201. package/lib/dist/lib/portConfig.js.map +1 -0
  202. package/lib/dist/lib/preferences.d.ts +63 -0
  203. package/lib/dist/lib/preferences.d.ts.map +1 -0
  204. package/lib/dist/lib/preferences.js +117 -0
  205. package/lib/dist/lib/preferences.js.map +1 -0
  206. package/lib/dist/lib/resolveAgentModel.d.ts +22 -0
  207. package/lib/dist/lib/resolveAgentModel.d.ts.map +1 -0
  208. package/lib/dist/lib/resolveAgentModel.js +37 -0
  209. package/lib/dist/lib/resolveAgentModel.js.map +1 -0
  210. package/lib/dist/lib/runStats.d.ts +116 -0
  211. package/lib/dist/lib/runStats.d.ts.map +1 -0
  212. package/lib/dist/lib/runStats.js +192 -0
  213. package/lib/dist/lib/runStats.js.map +1 -0
  214. package/lib/dist/lib/telemetry/constants.d.ts +60 -0
  215. package/lib/dist/lib/telemetry/constants.d.ts.map +1 -0
  216. package/lib/dist/lib/telemetry/constants.js +87 -0
  217. package/lib/dist/lib/telemetry/constants.js.map +1 -0
  218. package/lib/dist/lib/telemetry/evalSpans.d.ts +61 -0
  219. package/lib/dist/lib/telemetry/evalSpans.d.ts.map +1 -0
  220. package/lib/dist/lib/telemetry/evalSpans.js +254 -0
  221. package/lib/dist/lib/telemetry/evalSpans.js.map +1 -0
  222. package/lib/dist/lib/telemetry/index.d.ts +11 -0
  223. package/lib/dist/lib/telemetry/index.d.ts.map +1 -0
  224. package/lib/dist/lib/telemetry/index.js +15 -0
  225. package/lib/dist/lib/telemetry/index.js.map +1 -0
  226. package/lib/dist/lib/telemetry/opensearchExporter.d.ts +43 -0
  227. package/lib/dist/lib/telemetry/opensearchExporter.d.ts.map +1 -0
  228. package/lib/dist/lib/telemetry/opensearchExporter.js +217 -0
  229. package/lib/dist/lib/telemetry/opensearchExporter.js.map +1 -0
  230. package/lib/dist/lib/telemetry/provider.d.ts +55 -0
  231. package/lib/dist/lib/telemetry/provider.d.ts.map +1 -0
  232. package/lib/dist/lib/telemetry/provider.js +140 -0
  233. package/lib/dist/lib/telemetry/provider.js.map +1 -0
  234. package/lib/dist/lib/testCaseLabels.d.ts +34 -0
  235. package/lib/dist/lib/testCaseLabels.d.ts.map +1 -0
  236. package/lib/dist/lib/testCaseLabels.js +88 -0
  237. package/lib/dist/lib/testCaseLabels.js.map +1 -0
  238. package/lib/dist/lib/testCaseValidation.d.ts +140 -0
  239. package/lib/dist/lib/testCaseValidation.d.ts.map +1 -0
  240. package/lib/dist/lib/testCaseValidation.js +162 -0
  241. package/lib/dist/lib/testCaseValidation.js.map +1 -0
  242. package/lib/dist/lib/testCases/agentFixture.d.ts +80 -0
  243. package/lib/dist/lib/testCases/agentFixture.d.ts.map +1 -0
  244. package/lib/dist/lib/testCases/agentFixture.js +43 -0
  245. package/lib/dist/lib/testCases/agentFixture.js.map +1 -0
  246. package/lib/dist/lib/testCases/authoringSurface.d.ts +10 -0
  247. package/lib/dist/lib/testCases/authoringSurface.d.ts.map +1 -0
  248. package/lib/dist/lib/testCases/authoringSurface.js +54 -0
  249. package/lib/dist/lib/testCases/authoringSurface.js.map +1 -0
  250. package/lib/dist/lib/testCases/codemod.d.ts +13 -0
  251. package/lib/dist/lib/testCases/codemod.d.ts.map +1 -0
  252. package/lib/dist/lib/testCases/codemod.js +169 -0
  253. package/lib/dist/lib/testCases/codemod.js.map +1 -0
  254. package/lib/dist/lib/testCases/define.d.ts +114 -0
  255. package/lib/dist/lib/testCases/define.d.ts.map +1 -0
  256. package/lib/dist/lib/testCases/define.js +253 -0
  257. package/lib/dist/lib/testCases/define.js.map +1 -0
  258. package/lib/dist/lib/testCases/evaluators.d.ts +80 -0
  259. package/lib/dist/lib/testCases/evaluators.d.ts.map +1 -0
  260. package/lib/dist/lib/testCases/evaluators.js +105 -0
  261. package/lib/dist/lib/testCases/evaluators.js.map +1 -0
  262. package/lib/dist/lib/testCases/index.d.ts +14 -0
  263. package/lib/dist/lib/testCases/index.d.ts.map +1 -0
  264. package/lib/dist/lib/testCases/index.js +12 -0
  265. package/lib/dist/lib/testCases/index.js.map +1 -0
  266. package/lib/dist/lib/testCases/judge.d.ts +165 -0
  267. package/lib/dist/lib/testCases/judge.d.ts.map +1 -0
  268. package/lib/dist/lib/testCases/judge.js +359 -0
  269. package/lib/dist/lib/testCases/judge.js.map +1 -0
  270. package/lib/dist/lib/testCases/loader.d.ts +36 -0
  271. package/lib/dist/lib/testCases/loader.d.ts.map +1 -0
  272. package/lib/dist/lib/testCases/loader.js +179 -0
  273. package/lib/dist/lib/testCases/loader.js.map +1 -0
  274. package/lib/dist/lib/testCases/types.d.ts +242 -0
  275. package/lib/dist/lib/testCases/types.d.ts.map +1 -0
  276. package/lib/dist/lib/testCases/types.js +6 -0
  277. package/lib/dist/lib/testCases/types.js.map +1 -0
  278. package/lib/dist/lib/theme.d.ts +6 -0
  279. package/lib/dist/lib/theme.d.ts.map +1 -0
  280. package/lib/dist/lib/theme.js +36 -0
  281. package/lib/dist/lib/theme.js.map +1 -0
  282. package/lib/dist/lib/uiTelemetry.d.ts +7 -0
  283. package/lib/dist/lib/uiTelemetry.d.ts.map +1 -0
  284. package/lib/dist/lib/uiTelemetry.js +25 -0
  285. package/lib/dist/lib/uiTelemetry.js.map +1 -0
  286. package/lib/dist/lib/utils.d.ts +111 -0
  287. package/lib/dist/lib/utils.d.ts.map +1 -0
  288. package/lib/dist/lib/utils.js +254 -0
  289. package/lib/dist/lib/utils.js.map +1 -0
  290. package/lib/dist/lib/workflow/consolidate.d.ts +12 -0
  291. package/lib/dist/lib/workflow/consolidate.d.ts.map +1 -0
  292. package/lib/dist/lib/workflow/consolidate.js +33 -0
  293. package/lib/dist/lib/workflow/consolidate.js.map +1 -0
  294. package/lib/dist/lib/workflow/index.d.ts +13 -0
  295. package/lib/dist/lib/workflow/index.d.ts.map +1 -0
  296. package/lib/dist/lib/workflow/index.js +12 -0
  297. package/lib/dist/lib/workflow/index.js.map +1 -0
  298. package/lib/dist/lib/workflow/ledger.d.ts +30 -0
  299. package/lib/dist/lib/workflow/ledger.d.ts.map +1 -0
  300. package/lib/dist/lib/workflow/ledger.js +41 -0
  301. package/lib/dist/lib/workflow/ledger.js.map +1 -0
  302. package/lib/dist/lib/workflow/pool.d.ts +13 -0
  303. package/lib/dist/lib/workflow/pool.d.ts.map +1 -0
  304. package/lib/dist/lib/workflow/pool.js +44 -0
  305. package/lib/dist/lib/workflow/pool.js.map +1 -0
  306. package/lib/dist/lib/workflow/source.d.ts +22 -0
  307. package/lib/dist/lib/workflow/source.d.ts.map +1 -0
  308. package/lib/dist/lib/workflow/source.js +29 -0
  309. package/lib/dist/lib/workflow/source.js.map +1 -0
  310. package/lib/dist/lib/workflow/stepB.d.ts +71 -0
  311. package/lib/dist/lib/workflow/stepB.d.ts.map +1 -0
  312. package/lib/dist/lib/workflow/stepB.js +99 -0
  313. package/lib/dist/lib/workflow/stepB.js.map +1 -0
  314. package/lib/dist/lib/workflow/types.d.ts +86 -0
  315. package/lib/dist/lib/workflow/types.d.ts.map +1 -0
  316. package/lib/dist/lib/workflow/types.js +6 -0
  317. package/lib/dist/lib/workflow/types.js.map +1 -0
  318. package/lib/dist/lib/workflow/workflow.d.ts +119 -0
  319. package/lib/dist/lib/workflow/workflow.d.ts.map +1 -0
  320. package/lib/dist/lib/workflow/workflow.js +195 -0
  321. package/lib/dist/lib/workflow/workflow.js.map +1 -0
  322. package/lib/dist/services/agent/aguiConverter.d.ts +50 -0
  323. package/lib/dist/services/agent/aguiConverter.d.ts.map +1 -0
  324. package/lib/dist/services/agent/aguiConverter.js +449 -0
  325. package/lib/dist/services/agent/aguiConverter.js.map +1 -0
  326. package/lib/dist/services/agent/index.d.ts +10 -0
  327. package/lib/dist/services/agent/index.d.ts.map +1 -0
  328. package/lib/dist/services/agent/index.js +12 -0
  329. package/lib/dist/services/agent/index.js.map +1 -0
  330. package/lib/dist/services/agent/payloadBuilder.d.ts +33 -0
  331. package/lib/dist/services/agent/payloadBuilder.d.ts.map +1 -0
  332. package/lib/dist/services/agent/payloadBuilder.js +75 -0
  333. package/lib/dist/services/agent/payloadBuilder.js.map +1 -0
  334. package/lib/dist/services/agent/sseStream.d.ts +43 -0
  335. package/lib/dist/services/agent/sseStream.d.ts.map +1 -0
  336. package/lib/dist/services/agent/sseStream.js +223 -0
  337. package/lib/dist/services/agent/sseStream.js.map +1 -0
  338. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.d.ts +44 -0
  339. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.d.ts.map +1 -0
  340. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.js +95 -0
  341. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.js.map +1 -0
  342. package/lib/dist/services/connectors/base/BaseConnector.d.ts +81 -0
  343. package/lib/dist/services/connectors/base/BaseConnector.d.ts.map +1 -0
  344. package/lib/dist/services/connectors/base/BaseConnector.js +170 -0
  345. package/lib/dist/services/connectors/base/BaseConnector.js.map +1 -0
  346. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +126 -0
  347. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -0
  348. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +417 -0
  349. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -0
  350. package/lib/dist/services/connectors/index.d.ts +13 -0
  351. package/lib/dist/services/connectors/index.d.ts.map +1 -0
  352. package/lib/dist/services/connectors/index.js +32 -0
  353. package/lib/dist/services/connectors/index.js.map +1 -0
  354. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +48 -0
  355. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -0
  356. package/lib/dist/services/connectors/kiro/KiroConnector.js +158 -0
  357. package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -0
  358. package/lib/dist/services/connectors/langgraph/LangGraphConnector.d.ts +36 -0
  359. package/lib/dist/services/connectors/langgraph/LangGraphConnector.d.ts.map +1 -0
  360. package/lib/dist/services/connectors/langgraph/LangGraphConnector.js +175 -0
  361. package/lib/dist/services/connectors/langgraph/LangGraphConnector.js.map +1 -0
  362. package/lib/dist/services/connectors/mock/MockConnector.d.ts +37 -0
  363. package/lib/dist/services/connectors/mock/MockConnector.d.ts.map +1 -0
  364. package/lib/dist/services/connectors/mock/MockConnector.js +120 -0
  365. package/lib/dist/services/connectors/mock/MockConnector.js.map +1 -0
  366. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.d.ts +42 -0
  367. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.d.ts.map +1 -0
  368. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.js +133 -0
  369. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.js.map +1 -0
  370. package/lib/dist/services/connectors/pi/PiConnector.d.ts +87 -0
  371. package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -0
  372. package/lib/dist/services/connectors/pi/PiConnector.js +274 -0
  373. package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -0
  374. package/lib/dist/services/connectors/registry.d.ts +57 -0
  375. package/lib/dist/services/connectors/registry.d.ts.map +1 -0
  376. package/lib/dist/services/connectors/registry.js +106 -0
  377. package/lib/dist/services/connectors/registry.js.map +1 -0
  378. package/lib/dist/services/connectors/rest/RESTConnector.d.ts +38 -0
  379. package/lib/dist/services/connectors/rest/RESTConnector.d.ts.map +1 -0
  380. package/lib/dist/services/connectors/rest/RESTConnector.js +117 -0
  381. package/lib/dist/services/connectors/rest/RESTConnector.js.map +1 -0
  382. package/lib/dist/services/connectors/server.d.ts +13 -0
  383. package/lib/dist/services/connectors/server.d.ts.map +1 -0
  384. package/lib/dist/services/connectors/server.js +34 -0
  385. package/lib/dist/services/connectors/server.js.map +1 -0
  386. package/lib/dist/services/connectors/strands/StrandsConnector.d.ts +48 -0
  387. package/lib/dist/services/connectors/strands/StrandsConnector.d.ts.map +1 -0
  388. package/lib/dist/services/connectors/strands/StrandsConnector.js +221 -0
  389. package/lib/dist/services/connectors/strands/StrandsConnector.js.map +1 -0
  390. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +88 -0
  391. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -0
  392. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +426 -0
  393. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -0
  394. package/lib/dist/services/connectors/types.d.ts +213 -0
  395. package/lib/dist/services/connectors/types.d.ts.map +1 -0
  396. package/lib/dist/services/connectors/types.js +6 -0
  397. package/lib/dist/services/connectors/types.js.map +1 -0
  398. package/lib/dist/services/evaluation/bedrockJudge.d.ts +71 -0
  399. package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -0
  400. package/lib/dist/services/evaluation/bedrockJudge.js +169 -0
  401. package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -0
  402. package/lib/dist/services/evaluation/evaluatorError.d.ts +56 -0
  403. package/lib/dist/services/evaluation/evaluatorError.d.ts.map +1 -0
  404. package/lib/dist/services/evaluation/evaluatorError.js +56 -0
  405. package/lib/dist/services/evaluation/evaluatorError.js.map +1 -0
  406. package/lib/dist/services/evaluation/index.d.ts +106 -0
  407. package/lib/dist/services/evaluation/index.d.ts.map +1 -0
  408. package/lib/dist/services/evaluation/index.js +684 -0
  409. package/lib/dist/services/evaluation/index.js.map +1 -0
  410. package/lib/dist/services/evaluation/mockTrajectory.d.ts +3 -0
  411. package/lib/dist/services/evaluation/mockTrajectory.d.ts.map +1 -0
  412. package/lib/dist/services/evaluation/mockTrajectory.js +72 -0
  413. package/lib/dist/services/evaluation/mockTrajectory.js.map +1 -0
  414. package/lib/dist/services/opensearch/client.d.ts +26 -0
  415. package/lib/dist/services/opensearch/client.d.ts.map +1 -0
  416. package/lib/dist/services/opensearch/client.js +131 -0
  417. package/lib/dist/services/opensearch/client.js.map +1 -0
  418. package/lib/dist/services/opensearch/index.d.ts +16 -0
  419. package/lib/dist/services/opensearch/index.d.ts.map +1 -0
  420. package/lib/dist/services/opensearch/index.js +25 -0
  421. package/lib/dist/services/opensearch/index.js.map +1 -0
  422. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +123 -0
  423. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -0
  424. package/lib/dist/services/storage/asyncBenchmarkStorage.js +440 -0
  425. package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -0
  426. package/lib/dist/services/storage/asyncRunStorage.d.ts +135 -0
  427. package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -0
  428. package/lib/dist/services/storage/asyncRunStorage.js +524 -0
  429. package/lib/dist/services/storage/asyncRunStorage.js.map +1 -0
  430. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +170 -0
  431. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -0
  432. package/lib/dist/services/storage/asyncTestCaseStorage.js +301 -0
  433. package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -0
  434. package/lib/dist/services/storage/index.d.ts +17 -0
  435. package/lib/dist/services/storage/index.d.ts.map +1 -0
  436. package/lib/dist/services/storage/index.js +20 -0
  437. package/lib/dist/services/storage/index.js.map +1 -0
  438. package/lib/dist/services/storage/migration.d.ts +54 -0
  439. package/lib/dist/services/storage/migration.d.ts.map +1 -0
  440. package/lib/dist/services/storage/migration.js +296 -0
  441. package/lib/dist/services/storage/migration.js.map +1 -0
  442. package/lib/dist/services/storage/opensearchClient.d.ts +947 -0
  443. package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -0
  444. package/lib/dist/services/storage/opensearchClient.js +442 -0
  445. package/lib/dist/services/storage/opensearchClient.js.map +1 -0
  446. package/lib/dist/services/traces/browserRecovery.d.ts +37 -0
  447. package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -0
  448. package/lib/dist/services/traces/browserRecovery.js +108 -0
  449. package/lib/dist/services/traces/browserRecovery.js.map +1 -0
  450. package/lib/dist/services/traces/categoryStyles.d.ts +21 -0
  451. package/lib/dist/services/traces/categoryStyles.d.ts.map +1 -0
  452. package/lib/dist/services/traces/categoryStyles.js +56 -0
  453. package/lib/dist/services/traces/categoryStyles.js.map +1 -0
  454. package/lib/dist/services/traces/executionOrderTransform.d.ts +35 -0
  455. package/lib/dist/services/traces/executionOrderTransform.d.ts.map +1 -0
  456. package/lib/dist/services/traces/executionOrderTransform.js +313 -0
  457. package/lib/dist/services/traces/executionOrderTransform.js.map +1 -0
  458. package/lib/dist/services/traces/fetchSpansForRun.d.ts +86 -0
  459. package/lib/dist/services/traces/fetchSpansForRun.d.ts.map +1 -0
  460. package/lib/dist/services/traces/fetchSpansForRun.js +69 -0
  461. package/lib/dist/services/traces/fetchSpansForRun.js.map +1 -0
  462. package/lib/dist/services/traces/flowTransform.d.ts +24 -0
  463. package/lib/dist/services/traces/flowTransform.d.ts.map +1 -0
  464. package/lib/dist/services/traces/flowTransform.js +228 -0
  465. package/lib/dist/services/traces/flowTransform.js.map +1 -0
  466. package/lib/dist/services/traces/index.d.ts +121 -0
  467. package/lib/dist/services/traces/index.d.ts.map +1 -0
  468. package/lib/dist/services/traces/index.js +255 -0
  469. package/lib/dist/services/traces/index.js.map +1 -0
  470. package/lib/dist/services/traces/intentTransform.d.ts +20 -0
  471. package/lib/dist/services/traces/intentTransform.d.ts.map +1 -0
  472. package/lib/dist/services/traces/intentTransform.js +131 -0
  473. package/lib/dist/services/traces/intentTransform.js.map +1 -0
  474. package/lib/dist/services/traces/judgeAgentsHints.d.ts +63 -0
  475. package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -0
  476. package/lib/dist/services/traces/judgeAgentsHints.js +89 -0
  477. package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -0
  478. package/lib/dist/services/traces/messageExtraction.d.ts +15 -0
  479. package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -0
  480. package/lib/dist/services/traces/messageExtraction.js +313 -0
  481. package/lib/dist/services/traces/messageExtraction.js.map +1 -0
  482. package/lib/dist/services/traces/spanCategorization.d.ts +63 -0
  483. package/lib/dist/services/traces/spanCategorization.d.ts.map +1 -0
  484. package/lib/dist/services/traces/spanCategorization.js +276 -0
  485. package/lib/dist/services/traces/spanCategorization.js.map +1 -0
  486. package/lib/dist/services/traces/spanPreprocessing.d.ts +37 -0
  487. package/lib/dist/services/traces/spanPreprocessing.d.ts.map +1 -0
  488. package/lib/dist/services/traces/spanPreprocessing.js +102 -0
  489. package/lib/dist/services/traces/spanPreprocessing.js.map +1 -0
  490. package/lib/dist/services/traces/spansToTrajectory.d.ts +36 -0
  491. package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -0
  492. package/lib/dist/services/traces/spansToTrajectory.js +425 -0
  493. package/lib/dist/services/traces/spansToTrajectory.js.map +1 -0
  494. package/lib/dist/services/traces/toolSimilarity.d.ts +35 -0
  495. package/lib/dist/services/traces/toolSimilarity.d.ts.map +1 -0
  496. package/lib/dist/services/traces/toolSimilarity.js +203 -0
  497. package/lib/dist/services/traces/toolSimilarity.js.map +1 -0
  498. package/lib/dist/services/traces/traceComparison.d.ts +31 -0
  499. package/lib/dist/services/traces/traceComparison.d.ts.map +1 -0
  500. package/lib/dist/services/traces/traceComparison.js +318 -0
  501. package/lib/dist/services/traces/traceComparison.js.map +1 -0
  502. package/lib/dist/services/traces/traceGrouping.d.ts +19 -0
  503. package/lib/dist/services/traces/traceGrouping.d.ts.map +1 -0
  504. package/lib/dist/services/traces/traceGrouping.js +107 -0
  505. package/lib/dist/services/traces/traceGrouping.js.map +1 -0
  506. package/lib/dist/services/traces/tracePoller.d.ts +108 -0
  507. package/lib/dist/services/traces/tracePoller.d.ts.map +1 -0
  508. package/lib/dist/services/traces/tracePoller.js +475 -0
  509. package/lib/dist/services/traces/tracePoller.js.map +1 -0
  510. package/lib/dist/services/traces/traceStats.d.ts +45 -0
  511. package/lib/dist/services/traces/traceStats.d.ts.map +1 -0
  512. package/lib/dist/services/traces/traceStats.js +114 -0
  513. package/lib/dist/services/traces/traceStats.js.map +1 -0
  514. package/lib/dist/services/traces/traceSummary.d.ts +47 -0
  515. package/lib/dist/services/traces/traceSummary.d.ts.map +1 -0
  516. package/lib/dist/services/traces/traceSummary.js +68 -0
  517. package/lib/dist/services/traces/traceSummary.js.map +1 -0
  518. package/lib/dist/services/traces/utils.d.ts +33 -0
  519. package/lib/dist/services/traces/utils.d.ts.map +1 -0
  520. package/lib/dist/services/traces/utils.js +114 -0
  521. package/lib/dist/services/traces/utils.js.map +1 -0
  522. package/lib/dist/types/agui.d.ts +13 -0
  523. package/lib/dist/types/agui.d.ts.map +1 -0
  524. package/lib/dist/types/agui.js +16 -0
  525. package/lib/dist/types/agui.js.map +1 -0
  526. package/lib/dist/types/index.d.ts +1229 -0
  527. package/lib/dist/types/index.d.ts.map +1 -0
  528. package/lib/dist/types/index.js +12 -0
  529. package/lib/dist/types/index.js.map +1 -0
  530. package/lib/dist/types/skills.d.ts +146 -0
  531. package/lib/dist/types/skills.d.ts.map +1 -0
  532. package/lib/dist/types/skills.js +6 -0
  533. package/lib/dist/types/skills.js.map +1 -0
  534. package/observio-sample-agent/pi-package/README.md +112 -0
  535. package/observio-sample-agent/pi-package/extensions/agent-health.ts +373 -0
  536. package/observio-sample-agent/pi-package/package.json +17 -0
  537. package/observio-sample-agent/pi-package/prompts/agent-health.md +37 -0
  538. package/observio-sample-agent/pi-package/skills/create-pr/SKILL.md +88 -0
  539. package/observio-sample-agent/pi-package/skills/fix-bug/SKILL.md +71 -0
  540. package/observio-sample-agent/pi-package/skills/implement-feature/SKILL.md +156 -0
  541. package/observio-sample-agent/pi-package/skills/instrument-otel/SKILL.md +208 -0
  542. package/observio-sample-agent/pi-package/skills/setup-collector/SKILL.md +146 -0
  543. package/observio-sample-agent/pi-package/skills/write-test/SKILL.md +115 -0
  544. package/package.json +45 -9
  545. package/server/dist/app.js +34031 -18731
  546. package/server/dist/index.js +30737 -15338
  547. package/tsconfig.lib.json +71 -0
  548. package/dist/assets/index-BOIP5L7h.js +0 -246
  549. package/dist/assets/index-CU9YKpAL.css +0 -1
  550. package/lib/dist/config/index.js +0 -452
  551. package/lib/dist/index.js +0 -1725
@@ -0,0 +1,1229 @@
1
+ import type { Node, Edge } from '@xyflow/react';
2
+ export type Difficulty = 'Easy' | 'Medium' | 'Hard';
3
+ export type DateFormatVariant = 'date' | 'datetime' | 'detailed';
4
+ export type JudgeProvider = 'demo' | 'bedrock' | 'openai-compatible' | 'litellm' | 'claude-code' | 'agentic' | 'pi' | 'agent';
5
+ export interface AssistantMessage {
6
+ role: 'user' | 'assistant';
7
+ content: string;
8
+ timestamp: string;
9
+ }
10
+ export interface AssistantContext {
11
+ currentUrl?: string;
12
+ benchmarkId?: string;
13
+ runId?: string;
14
+ traceId?: string;
15
+ testCaseId?: string;
16
+ /**
17
+ * On comparison pages (`/compare/:benchmarkId?runs=a,b,…`), the list of run
18
+ * IDs the user is currently comparing. The assistant pre-loads these into
19
+ * the grounded snapshot so it can answer cross-run questions even before
20
+ * reaching for tools.
21
+ */
22
+ comparisonRunIds?: string[];
23
+ }
24
+ export type ConnectorProtocol = 'agui-streaming' | 'rest' | 'openai-compatible' | 'subprocess' | 'claude-code' | 'pi' | 'strands' | 'langgraph' | 'mock';
25
+ export interface ModelConfig {
26
+ model_id: string;
27
+ display_name: string;
28
+ provider: JudgeProvider;
29
+ context_window: number;
30
+ max_output_tokens: number;
31
+ }
32
+ export interface BeforeRequestContext {
33
+ endpoint: string;
34
+ payload: any;
35
+ headers: Record<string, string>;
36
+ }
37
+ export interface AfterResponseContext {
38
+ response: any;
39
+ trajectory: TrajectoryStep[];
40
+ runId?: string;
41
+ /** Full array of raw events from the connector (protocol-specific) */
42
+ rawEvents?: any[];
43
+ /** Connector metadata (e.g., threadId, sessionId, exitCode) */
44
+ metadata?: Record<string, any>;
45
+ }
46
+ export interface BuildTrajectoryContext {
47
+ spans: Span[];
48
+ runId: string;
49
+ }
50
+ /**
51
+ * Context passed to a custom judge hook.
52
+ * Contains all data needed for evaluation: trajectory, traces, and expected outcomes.
53
+ * Also provides fetchTraces as an SDK utility for additional trace fetching.
54
+ */
55
+ export interface JudgeContext {
56
+ trajectory: TrajectoryStep[];
57
+ traces: Span[];
58
+ expectedOutcomes: string[];
59
+ expectedTrajectory?: string[];
60
+ runId: string;
61
+ /** SDK utility: fetch traces by run IDs from OpenSearch */
62
+ fetchTraces: (runIds: string[]) => Promise<{
63
+ spans: Span[];
64
+ }>;
65
+ }
66
+ /**
67
+ * Result returned by a custom judge hook.
68
+ */
69
+ export interface JudgeResult {
70
+ passFailStatus: 'passed' | 'failed';
71
+ metrics: {
72
+ accuracy: number;
73
+ faithfulness?: number;
74
+ latency_score?: number;
75
+ trajectory_alignment_score?: number;
76
+ [key: string]: number | undefined;
77
+ };
78
+ llmJudgeReasoning: string;
79
+ improvementStrategies?: string[];
80
+ }
81
+ export interface AgentHooks {
82
+ /**
83
+ * Called before sending request to agent.
84
+ * Use to modify endpoint, payload, or headers.
85
+ */
86
+ beforeRequest?: (context: BeforeRequestContext) => Promise<BeforeRequestContext>;
87
+ /**
88
+ * Called after receiving response from agent.
89
+ * Use to extract runId from custom response formats (e.g., PER memory_id).
90
+ */
91
+ afterResponse?: (context: AfterResponseContext) => Promise<AfterResponseContext>;
92
+ /**
93
+ * Called when building trajectory from OTEL traces.
94
+ * Use to customize trajectory extraction for agents with custom span formats.
95
+ */
96
+ buildTrajectory?: (context: BuildTrajectoryContext) => Promise<TrajectoryStep[]>;
97
+ /**
98
+ * Custom judge hook. When defined, replaces the built-in Bedrock judge.
99
+ * Receives trajectory + traces + expected outcomes, returns pass/fail evaluation.
100
+ *
101
+ * @example
102
+ * ```typescript
103
+ * hooks: {
104
+ * judge: async ({ trajectory, traces, expectedOutcomes, fetchTraces }) => {
105
+ * // Custom evaluation logic using traces
106
+ * const relevantSpans = traces.filter(s => s.attributes?.['gen_ai.system']);
107
+ * return {
108
+ * passFailStatus: relevantSpans.length > 0 ? 'passed' : 'failed',
109
+ * metrics: { accuracy: 85 },
110
+ * llmJudgeReasoning: 'Custom evaluation based on trace analysis',
111
+ * };
112
+ * }
113
+ * }
114
+ * ```
115
+ */
116
+ judge?: (context: JudgeContext) => Promise<JudgeResult>;
117
+ }
118
+ export interface AgentConfig {
119
+ key: string;
120
+ name: string;
121
+ endpoint: string;
122
+ description?: string;
123
+ enabled?: boolean;
124
+ headers?: Record<string, string>;
125
+ auth?: ConnectorAuthConfig;
126
+ useTraces?: boolean;
127
+ /**
128
+ * Configurable trace polling settings (used when `useTraces: true`).
129
+ *
130
+ * Two distinct polling paths honour these values, with different defaults
131
+ * because they have different ergonomic constraints:
132
+ *
133
+ * - **Judge poller** (`services/traces/tracePoller.ts`, runs in the
134
+ * background after the agent finishes, before the LLM judge fires)
135
+ * defaults to `intervalMs: 10000` and `maxAttempts: 60` — a 10-minute
136
+ * total budget that's fine because the user already sees a "pending"
137
+ * badge while it polls.
138
+ * - **SDK pre-load** (`services/traces/fetchSpansForRun.ts`, runs
139
+ * synchronously inside a deterministic test body before the body's
140
+ * first assertion) defaults to `intervalMs: 1000` and `maxAttempts:
141
+ * 10` — a ~10-second total budget so the test isn't blocked.
142
+ *
143
+ * Both paths additionally honour `TRACE_POLL_INTERVAL_MS` and
144
+ * `TRACE_POLL_MAX_ATTEMPTS` env vars (the env vars override the
145
+ * code defaults), and both enforce a hard ceiling of 60 attempts so
146
+ * a misconfigured agent can't lock a test for an unbounded time.
147
+ *
148
+ * Setting either field on this object overrides the path's own default
149
+ * for that specific agent on both paths.
150
+ */
151
+ tracePolling?: {
152
+ intervalMs?: number;
153
+ maxAttempts?: number;
154
+ };
155
+ /**
156
+ * OTel `service.name` resource attribute that this agent reports under.
157
+ * Defaults to {@link AgentConfig.key} when not set, which is correct for
158
+ * agents whose OTel SDK uses the same identifier as the config key (e.g.
159
+ * `claude-code`). Override only when the agent's OTel service name differs
160
+ * from its config key, e.g. `observio` -> `observio-sample-agent`.
161
+ *
162
+ * Used by the Agent Traces page to translate the user's cross-page agent
163
+ * filter (`agent-health:prefs:agentFilter`, which stores agent keys) into
164
+ * the actual `service.name` to filter by in OpenSearch queries.
165
+ */
166
+ traceServiceName?: string;
167
+ connectorType?: ConnectorProtocol;
168
+ connectorConfig?: Record<string, any>;
169
+ hooks?: AgentHooks;
170
+ isCustom?: boolean;
171
+ builtIn?: boolean;
172
+ }
173
+ /**
174
+ * Authentication config for agents (serializable subset of ConnectorAuth).
175
+ * Used in AgentConfig for config files — avoids importing connector types.
176
+ */
177
+ export interface ConnectorAuthConfig {
178
+ type: 'none' | 'basic' | 'bearer' | 'api-key' | 'aws-sigv4';
179
+ username?: string;
180
+ password?: string;
181
+ token?: string;
182
+ awsRegion?: string;
183
+ awsService?: string;
184
+ headers?: Record<string, string>;
185
+ }
186
+ export interface AppConfig {
187
+ agents: AgentConfig[];
188
+ models: Record<string, ModelConfig>;
189
+ defaults: {
190
+ retry_attempts: number;
191
+ retry_delay_ms: number;
192
+ };
193
+ }
194
+ export declare enum ToolCallStatus {
195
+ SUCCESS = "SUCCESS",
196
+ FAILURE = "FAILURE"
197
+ }
198
+ export interface TrajectoryStep {
199
+ id: string;
200
+ timestamp: number;
201
+ type: 'tool_result' | 'assistant' | 'action' | 'response' | 'thinking';
202
+ content: string;
203
+ toolName?: string;
204
+ toolArgs?: Record<string, any>;
205
+ toolOutput?: any;
206
+ status?: ToolCallStatus;
207
+ latencyMs?: number;
208
+ }
209
+ export interface EvaluationMetrics {
210
+ accuracy?: number;
211
+ faithfulness?: number;
212
+ latency_score?: number;
213
+ trajectory_alignment_score?: number;
214
+ [key: string]: number | undefined;
215
+ }
216
+ export interface ImprovementStrategy {
217
+ category: string;
218
+ issue: string;
219
+ recommendation: string;
220
+ priority: 'high' | 'medium' | 'low';
221
+ }
222
+ /**
223
+ * Scoring metric definition for evaluators
224
+ */
225
+ export interface ScoringMetric {
226
+ name: string;
227
+ description?: string;
228
+ weight: number;
229
+ scale: number;
230
+ }
231
+ /**
232
+ * Scoring configuration for an evaluator
233
+ */
234
+ export interface ScoringConfig {
235
+ metrics: ScoringMetric[];
236
+ passThreshold: number;
237
+ scale: number;
238
+ }
239
+ /**
240
+ * Inference configuration for an evaluator
241
+ */
242
+ export interface InferenceConfig {
243
+ provider?: JudgeProvider;
244
+ modelId?: string;
245
+ temperature?: number;
246
+ maxTokens?: number;
247
+ }
248
+ /**
249
+ * Evaluator version - immutable snapshot of evaluator configuration
250
+ */
251
+ export interface EvaluatorVersion {
252
+ version: number;
253
+ createdAt: string;
254
+ systemPrompt: string;
255
+ scoringConfig: ScoringConfig;
256
+ inferenceConfig: InferenceConfig;
257
+ }
258
+ /**
259
+ * Evaluator - pluggable judge configuration
260
+ * Defines how agent performance is evaluated
261
+ */
262
+ export interface Evaluator {
263
+ id: string;
264
+ name: string;
265
+ description: string;
266
+ isSystem: boolean;
267
+ tags?: string[];
268
+ currentVersion: number;
269
+ versions: EvaluatorVersion[];
270
+ createdAt: string;
271
+ updatedAt: string;
272
+ author?: string;
273
+ systemPrompt: string;
274
+ scoringConfig: ScoringConfig;
275
+ inferenceConfig: InferenceConfig;
276
+ }
277
+ export type PassFailStatus = 'passed' | 'failed';
278
+ export interface LLMJudgeResponse {
279
+ modelId: string;
280
+ timestamp: string;
281
+ promptTokens: number;
282
+ completionTokens: number;
283
+ latencyMs: number;
284
+ /**
285
+ * Raw judge text exactly as the model returned it (pre-JSON-parse). Set
286
+ * by the routing layer from `JudgeResponse.rawResponse`. Older callers
287
+ * stuffed the parsed `llmJudgeReasoning` into this field as a fallback;
288
+ * post evaluator-prompt-plumbing the field carries the actual unparsed
289
+ * model output for debugging "prompt edited but output didn't change"
290
+ * scenarios.
291
+ */
292
+ rawResponse: string;
293
+ /**
294
+ * Parsed numeric metrics. Open-ended `[key: string]: number` so a saved
295
+ * evaluator can declare arbitrary metric names in its `scoringConfig.metrics`
296
+ * and they flow through here unchanged. Legacy keys (`accuracy`,
297
+ * `faithfulness`, `latency_score`, `trajectory_alignment_score`) remain
298
+ * conventional but are no longer required — evaluators are pluggable.
299
+ */
300
+ parsedMetrics?: {
301
+ [key: string]: number | undefined;
302
+ };
303
+ improvementStrategies?: ImprovementStrategy[];
304
+ error?: string;
305
+ /**
306
+ * Any JSON keys the judge emitted that did NOT map onto a typed wire
307
+ * field or a declared metric. Captured by
308
+ * {@link parseJudgeResponse} (server/services/judgeResponseParser) so the
309
+ * run-detail "Judge debug" surface can show prompt-iteration output (e.g.
310
+ * `improvement_candidates`, `failure_tags`, `confidence`) without a code
311
+ * change. Empty/undefined when the model emitted only typed fields.
312
+ */
313
+ extraFields?: Record<string, unknown>;
314
+ /**
315
+ * Optional debug breadcrumbs persisted when `AH_JUDGE_DEBUG=1` (or in dev
316
+ * mode). Captures exactly what the run-detail UI needs to confirm "the
317
+ * prompt I saved is the prompt that ran" — the system prompt the model
318
+ * received, the user prompt, and which provider executed the call. The
319
+ * raw response itself is on the parent {@link rawResponse}.
320
+ *
321
+ * Disabled by default to keep persisted run docs lean (system prompts
322
+ * can be 10–20 KB).
323
+ */
324
+ judgeDebug?: {
325
+ /** Provider that executed the call: 'bedrock' | 'claude-code' | 'pi' | 'agent' | 'agentic' | 'openai-compatible' | 'litellm'. */
326
+ provider?: string;
327
+ /** Effective model id passed to the provider (post-resolution). */
328
+ modelId?: string;
329
+ /** Evaluator id used (system or user). */
330
+ evaluatorId?: string;
331
+ /** The full system prompt the model received. */
332
+ systemPrompt?: string;
333
+ /** The user-message prompt the model received. */
334
+ userPrompt?: string;
335
+ };
336
+ }
337
+ export interface RunAnnotation {
338
+ id: string;
339
+ reportId: string;
340
+ text: string;
341
+ timestamp: string;
342
+ tags?: string[];
343
+ author?: string;
344
+ }
345
+ /**
346
+ * @experimental Generic sidecar metadata for coding agent sessions.
347
+ * One document per session — stores annotations, status, tags, or any
348
+ * user-defined fields. The shape is intentionally open so callers can
349
+ * store whatever debug/analysis data they need.
350
+ */
351
+ export interface SessionMetadata {
352
+ agentKind: string;
353
+ sessionId: string;
354
+ /** Open-ended — callers define the schema. */
355
+ [key: string]: unknown;
356
+ }
357
+ export type MetricsStatus = 'pending' | 'calculating' | 'ready' | 'error';
358
+ export interface TestCaseRun {
359
+ id: string;
360
+ timestamp: string;
361
+ /**
362
+ * Human-readable name for this run (e.g. "Baseline", "Claude_02").
363
+ * Set from the user-supplied value in the run config dialog, or auto-generated
364
+ * server-side as `Run <short-id>` if not provided. Optional for backwards
365
+ * compatibility with runs created before the field existed — UI consumers
366
+ * should fall back to a generated label (see `getRunDisplayName`).
367
+ */
368
+ name?: string;
369
+ /** Optional human-readable description of what this run was testing. */
370
+ description?: string;
371
+ testCaseId: string;
372
+ testCaseVersion?: number;
373
+ experimentId?: string;
374
+ experimentRunId?: string;
375
+ agentName: string;
376
+ agentKey?: string;
377
+ modelName: string;
378
+ modelId?: string;
379
+ /**
380
+ * Optional judge model id, separate from {@link modelId} (which is the
381
+ * agent's LLM). Set explicitly via the run config (UI dropdown / CLI
382
+ * `--judge-model` / API `judgeModelId` field) or left unset to fall back
383
+ * to the evaluator's `inferenceConfig.modelId`, then the server-default
384
+ * Bedrock judge model. For agentic providers (`pi`, `agent`, `agentic`,
385
+ * `claude-code`) the value is informational — the provider picks its own
386
+ * model from its credentialed registry. Stored on the run document so the
387
+ * "Judge debug" surface and audit trail show which judge model was used.
388
+ */
389
+ judgeModelId?: string;
390
+ agentEndpoint?: string;
391
+ evaluatorId?: string;
392
+ status: 'running' | 'completed' | 'failed';
393
+ passFailStatus?: PassFailStatus;
394
+ trajectory: TrajectoryStep[];
395
+ metrics: EvaluationMetrics;
396
+ /**
397
+ * @deprecated Use `getJudgeReasoningText(report)` /
398
+ * `getJudgeMatcherResults(report)` from `lib/matchers/judgeAccessor`.
399
+ * The canonical judge surface is now `matcherResults[]` with
400
+ * `method: 'llm-judge'`. This flat-string field is kept as an
401
+ * Option-B backward-compat shim — it carries the most recent judge
402
+ * reasoning so old direct readers keep working, but new code MUST
403
+ * use the accessor.
404
+ */
405
+ llmJudgeReasoning: string;
406
+ improvementStrategies?: ImprovementStrategy[];
407
+ llmJudgeResponse?: LLMJudgeResponse;
408
+ /**
409
+ * W3C OTel trace id (32 hex). Stamped onto the run document at save time
410
+ * when polled spans expose one, used as the strongest correlation key for
411
+ * the run-detail Traces tab and the agent (trace) judge's `query_spans`
412
+ * tool. See #190 (this field as a top-level shortcut over re-extracting
413
+ * from `spans[0]`) and #264 (unified trace correlation strategies).
414
+ *
415
+ * Distinct from {@link runId} (the connector's run id, e.g.
416
+ * `subprocess-<timestamp>`); pre-fix the runner mis-stamped runId here
417
+ * which broke `traceId`-based queries.
418
+ */
419
+ traceId?: string;
420
+ openSearchLogs?: OpenSearchLog[];
421
+ annotations?: RunAnnotation[];
422
+ runId?: string;
423
+ /**
424
+ * Agent-emitted session id (e.g. Claude Code stamps `session.id` on every
425
+ * span of a run). Captured from the connector result and used as a precise
426
+ * per-run trace correlator (Strategy D) for agents that emit it but don't
427
+ * propagate W3C context or tag our `agent_health.run.id`.
428
+ */
429
+ sessionId?: string;
430
+ logs?: OpenSearchLog[];
431
+ rawEvents?: any[];
432
+ connectorProtocol?: ConnectorProtocol;
433
+ matcherResults?: import('../lib/matchers/types.js').MatcherResult[];
434
+ performanceMetrics?: TestCasePerformanceMetrics;
435
+ metricsStatus?: MetricsStatus;
436
+ traceFetchAttempts?: number;
437
+ lastTraceFetchAt?: string;
438
+ traceError?: string;
439
+ spans?: Span[];
440
+ }
441
+ export type EvaluationReport = TestCaseRun;
442
+ export type ContextItemDisposition = 'prompt' | 'connector' | 'documentation';
443
+ export interface AgentContextItem {
444
+ description: string;
445
+ value: string;
446
+ /** How this item is consumed. Absent is equivalent to `prompt`. */
447
+ disposition?: ContextItemDisposition;
448
+ }
449
+ export interface AgentToolDefinition {
450
+ name: string;
451
+ description: string;
452
+ parameters: {
453
+ type: 'object';
454
+ properties: Record<string, {
455
+ type: string;
456
+ description: string;
457
+ enum?: string[];
458
+ }>;
459
+ required?: string[];
460
+ };
461
+ }
462
+ export type Category = 'Baseline' | 'Smart Contextual Menu' | 'RCA' | 'Conversational Queries' | 'Top 10 Browsed Products' | 'Errors by Service' | 'Group by Error Type' | string;
463
+ export interface TestCaseVersion {
464
+ version: number;
465
+ createdAt: string;
466
+ initialPrompt?: string;
467
+ context: AgentContextItem[];
468
+ tools?: AgentToolDefinition[];
469
+ expectedPPL?: string;
470
+ expectedOutcomes?: string[];
471
+ expectedTrajectory?: {
472
+ step: number;
473
+ description: string;
474
+ requiredTools: string[];
475
+ }[];
476
+ followUpQuestions?: {
477
+ trigger: 'results_available' | 'error' | 'always';
478
+ question: string;
479
+ businessValue: string;
480
+ }[];
481
+ }
482
+ export interface TestCase {
483
+ id: string;
484
+ name: string;
485
+ description: string;
486
+ labels: string[];
487
+ /** @deprecated Use labels with 'category:' prefix instead */
488
+ category: Category;
489
+ /** @deprecated Use labels with 'subcategory:' prefix instead */
490
+ subcategory?: string;
491
+ /** @deprecated Use labels with 'difficulty:' prefix instead */
492
+ difficulty: Difficulty;
493
+ currentVersion: number;
494
+ versions: TestCaseVersion[];
495
+ sourceFile?: string;
496
+ sourceHash?: string;
497
+ sourceCode?: string;
498
+ sourceFileName?: string;
499
+ sourceLanguage?: 'javascript' | 'typescript';
500
+ isPromoted: boolean;
501
+ createdAt: string;
502
+ updatedAt: string;
503
+ lastRunAt?: string;
504
+ initialPrompt?: string;
505
+ context: AgentContextItem[];
506
+ tools?: AgentToolDefinition[];
507
+ expectedPPL?: string;
508
+ expectedOutcomes?: string[];
509
+ expectedTrajectory?: {
510
+ step: number;
511
+ description: string;
512
+ requiredTools: string[];
513
+ }[];
514
+ followUpQuestions?: {
515
+ trigger: 'results_available' | 'error' | 'always';
516
+ question: string;
517
+ businessValue: string;
518
+ }[];
519
+ }
520
+ export interface OpenSearchLog {
521
+ timestamp: string;
522
+ index: string;
523
+ message: string;
524
+ level?: string;
525
+ source?: string;
526
+ [key: string]: any;
527
+ }
528
+ export interface LogQueryParams {
529
+ startTime: Date;
530
+ endTime: Date;
531
+ size?: number;
532
+ query?: string;
533
+ }
534
+ export interface TraceMetrics {
535
+ runId: string;
536
+ traceId?: string;
537
+ inputTokens: number;
538
+ outputTokens: number;
539
+ totalTokens: number;
540
+ costUsd: number;
541
+ durationMs: number;
542
+ llmCalls: number;
543
+ toolCalls: number;
544
+ toolsUsed: string[];
545
+ status: 'success' | 'error' | 'pending';
546
+ }
547
+ export interface SpanEvent {
548
+ name: string;
549
+ time: string;
550
+ attributes?: Record<string, any>;
551
+ }
552
+ export interface Span {
553
+ traceId: string;
554
+ spanId: string;
555
+ parentSpanId?: string;
556
+ name: string;
557
+ startTime: string;
558
+ endTime: string;
559
+ duration?: number;
560
+ status: 'OK' | 'ERROR' | 'UNSET';
561
+ attributes?: Record<string, any>;
562
+ events?: SpanEvent[];
563
+ children?: Span[];
564
+ depth?: number;
565
+ hasChildren?: boolean;
566
+ }
567
+ export interface TimeRange {
568
+ startTime: number;
569
+ endTime: number;
570
+ duration: number;
571
+ }
572
+ export interface TraceQueryParams {
573
+ traceId?: string;
574
+ runIds?: string[];
575
+ sessionId?: string;
576
+ startTime?: number;
577
+ endTime?: number;
578
+ size?: number;
579
+ serviceName?: string;
580
+ textSearch?: string;
581
+ cursor?: string;
582
+ /**
583
+ * Strategy C (opt-in): include any spans where `serviceName` matches AND
584
+ * `startTime` falls within `[startedAt, endedAt]`. Used by the run-report
585
+ * Traces tab as a fallback for agents that don't propagate W3C trace context
586
+ * (TRACEPARENT) and don't tag spans with `gen_ai.request.id` matching our
587
+ * runId. May surface unrelated spans (concurrent runs, cross-team noise).
588
+ * See AGENTS.md → Trace correlation conventions.
589
+ *
590
+ * Strategy D: when `sessionId` is set on an entry, correlate precisely on
591
+ * `attributes.session.id` (unioned with the service.name + window fallback).
592
+ */
593
+ agents?: Array<{
594
+ serviceName: string;
595
+ startedAt: number;
596
+ endedAt: number;
597
+ sessionId?: string;
598
+ }>;
599
+ }
600
+ export interface ConversationMessage {
601
+ id: string;
602
+ timestamp: string;
603
+ role: 'user' | 'assistant' | 'tool_call' | 'tool_result' | 'system';
604
+ content: string;
605
+ metadata?: {
606
+ spanId?: string;
607
+ spanName?: string;
608
+ toolName?: string;
609
+ model?: string;
610
+ inputTokens?: number;
611
+ outputTokens?: number;
612
+ durationMs?: number;
613
+ };
614
+ }
615
+ export interface TraceSearchResult {
616
+ spans: Span[];
617
+ total: number;
618
+ warning?: string;
619
+ warningCategory?: 'auth' | 'connection' | 'index_not_found' | 'not_configured' | 'unknown';
620
+ suggestion?: string;
621
+ nextCursor?: string | null;
622
+ hasMore?: boolean;
623
+ }
624
+ /**
625
+ * Summary of a single trace (grouped spans)
626
+ * Used for trace list display before selecting one for detailed view
627
+ */
628
+ export interface TraceSummary {
629
+ traceId: string;
630
+ serviceName: string;
631
+ spanCount: number;
632
+ rootSpanName: string;
633
+ startTime: string;
634
+ duration: number;
635
+ hasErrors: boolean;
636
+ hasEvalSpans?: boolean;
637
+ spans: Span[];
638
+ }
639
+ /**
640
+ * Span category based on OTel GenAI semantic conventions
641
+ * @see https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/
642
+ */
643
+ export type SpanCategory = 'AGENT' | 'LLM' | 'TOOL' | 'EVAL' | 'ERROR' | 'OTHER';
644
+ /**
645
+ * Extended span with category metadata for tree visualization
646
+ */
647
+ export interface CategorizedSpan extends Span {
648
+ category: SpanCategory;
649
+ categoryLabel: string;
650
+ categoryColor: string;
651
+ categoryIcon: string;
652
+ displayName: string;
653
+ }
654
+ /**
655
+ * Configuration for tool similarity grouping
656
+ */
657
+ export interface ToolSimilarityConfig {
658
+ /** Which tool arguments to use for determining "sameness" */
659
+ keyArguments: string[];
660
+ /** Whether grouping is enabled */
661
+ enabled: boolean;
662
+ }
663
+ /**
664
+ * Grouped tool spans for similarity view
665
+ */
666
+ export interface ToolGroup {
667
+ toolName: string;
668
+ keyArgsValues: Record<string, any>;
669
+ spans: CategorizedSpan[];
670
+ count: number;
671
+ totalDuration: number;
672
+ avgDuration: number;
673
+ }
674
+ /**
675
+ * Aligned span pair for tree comparison
676
+ */
677
+ export interface AlignedSpanPair {
678
+ type: 'matched' | 'added' | 'removed' | 'modified';
679
+ leftSpan?: CategorizedSpan;
680
+ rightSpan?: CategorizedSpan;
681
+ similarity?: number;
682
+ children?: AlignedSpanPair[];
683
+ }
684
+ /**
685
+ * Result of comparing two trace trees
686
+ */
687
+ export interface TraceComparisonResult {
688
+ alignedTree: AlignedSpanPair[];
689
+ stats: {
690
+ totalLeft: number;
691
+ totalRight: number;
692
+ matched: number;
693
+ added: number;
694
+ removed: number;
695
+ modified: number;
696
+ };
697
+ }
698
+ /**
699
+ * Data payload for span nodes in React Flow
700
+ * Index signature required for React Flow compatibility
701
+ */
702
+ export interface SpanNodeData extends Record<string, unknown> {
703
+ span: CategorizedSpan;
704
+ totalDuration: number;
705
+ }
706
+ /**
707
+ * Result of transforming spans to React Flow format
708
+ */
709
+ export interface FlowTransformResult {
710
+ nodes: Node<SpanNodeData>[];
711
+ edges: Edge[];
712
+ }
713
+ /**
714
+ * Options for flow transformation
715
+ */
716
+ export interface FlowTransformOptions {
717
+ direction?: 'TB' | 'LR';
718
+ mode?: 'hierarchy' | 'execution-order';
719
+ nodeWidth?: number;
720
+ nodeHeight?: number;
721
+ nodeSpacingX?: number;
722
+ nodeSpacingY?: number;
723
+ }
724
+ /**
725
+ * Group of spans detected as parallel execution
726
+ */
727
+ export interface ParallelGroup {
728
+ spans: CategorizedSpan[];
729
+ startTime: number;
730
+ endTime: number;
731
+ }
732
+ /**
733
+ * Result of checking OTEL GenAI semantic convention compliance
734
+ */
735
+ export interface OTelComplianceResult {
736
+ isCompliant: boolean;
737
+ missingAttributes: string[];
738
+ }
739
+ /**
740
+ * Compressed node for Intent view - represents one or more consecutive same-category spans
741
+ */
742
+ export interface IntentNode {
743
+ id: string;
744
+ category: SpanCategory;
745
+ spans: CategorizedSpan[];
746
+ count: number;
747
+ displayName: string;
748
+ subtitle: string;
749
+ hasWarnings: boolean;
750
+ executionOrder: number;
751
+ startIndex: number;
752
+ totalDuration: number;
753
+ }
754
+ /**
755
+ * Metadata about storage availability and data source
756
+ * Included in list responses to inform clients about data provenance
757
+ */
758
+ export interface StorageMetadata {
759
+ /** Whether storage backend is configured (env vars set) */
760
+ storageConfigured: boolean;
761
+ /** Whether storage backend was reachable on this request */
762
+ storageReachable: boolean;
763
+ /** Count of items from persistent storage */
764
+ realDataCount: number;
765
+ /** Count of items from built-in sample data */
766
+ sampleDataCount: number;
767
+ /** Whether sample/demo data was included in this response */
768
+ sampleDataIncluded?: boolean;
769
+ /** Optional warning messages (e.g., connection errors) */
770
+ warnings?: string[];
771
+ }
772
+ /**
773
+ * Generic list response wrapper with metadata
774
+ */
775
+ export interface ListResponse<T> {
776
+ data: T[];
777
+ total: number;
778
+ meta: StorageMetadata;
779
+ }
780
+ /** Server-side performance metrics for a single test case evaluation */
781
+ export interface TestCasePerformanceMetrics {
782
+ durationMs: number;
783
+ agentDurationMs: number;
784
+ judgeDurationMs?: number;
785
+ judgeAttempts?: number;
786
+ }
787
+ /** Server-side performance metrics for an entire benchmark run */
788
+ export interface RunPerformanceMetrics {
789
+ durationMs: number;
790
+ concurrency: number;
791
+ avgTestCaseDurationMs: number;
792
+ maxTestCaseDurationMs: number;
793
+ minTestCaseDurationMs: number;
794
+ }
795
+ export interface RunStats {
796
+ /** Number of test cases that passed (passFailStatus === 'passed') */
797
+ passed: number;
798
+ /** Number of test cases that failed (passFailStatus === 'failed' or execution failed) */
799
+ failed: number;
800
+ /** Number of test cases still pending (running, or report not yet available) */
801
+ pending: number;
802
+ /**
803
+ * Number of test cases where the *evaluator* could not produce a verdict
804
+ * (e.g. judge validation error, trace polling timeout, post-trace callback
805
+ * failed). Excluded from `passed` and `failed` so a misconfigured evaluator
806
+ * doesn't silently poison aggregate pass rates.
807
+ *
808
+ * Optional for backward-compat: older stored runs predate this field and
809
+ * read as 0.
810
+ */
811
+ errored?: number;
812
+ /** Total number of test cases in the run */
813
+ total: number;
814
+ }
815
+ export type RunResultStatus = 'pending' | 'running' | 'completed' | 'failed' | 'cancelled';
816
+ export type BenchmarkRunStatus = 'pending' | 'running' | 'completed' | 'failed' | 'cancelled';
817
+ export interface BenchmarkVersion {
818
+ version: number;
819
+ createdAt: string;
820
+ testCaseIds: string[];
821
+ }
822
+ export interface TestCaseSnapshot {
823
+ id: string;
824
+ version: number;
825
+ name: string;
826
+ }
827
+ export interface BenchmarkRun {
828
+ id: string;
829
+ name: string;
830
+ description?: string;
831
+ createdAt: string;
832
+ completedAt?: string;
833
+ status?: BenchmarkRunStatus;
834
+ error?: string;
835
+ agentKey: string;
836
+ agentEndpoint?: string;
837
+ modelId: string;
838
+ /**
839
+ * Optional judge model id, distinct from {@link modelId} (the agent's
840
+ * LLM). Customer input via the run config dialog / CLI `--judge-model` /
841
+ * API. Falls back to `evaluator.inferenceConfig.modelId`, then the
842
+ * server-default Bedrock judge model. Ignored by agentic providers
843
+ * (`pi`, `agent`, `agentic`, `claude-code`) which pick their own model.
844
+ */
845
+ judgeModelId?: string;
846
+ evaluatorId?: string;
847
+ headers?: Record<string, string>;
848
+ concurrency?: number;
849
+ benchmarkVersion?: number;
850
+ testCaseSnapshots?: TestCaseSnapshot[];
851
+ results: Record<string, {
852
+ reportId: string;
853
+ status: RunResultStatus;
854
+ error?: string;
855
+ performanceMetrics?: TestCasePerformanceMetrics;
856
+ }>;
857
+ stats?: RunStats;
858
+ performanceMetrics?: RunPerformanceMetrics;
859
+ }
860
+ export interface Benchmark {
861
+ id: string;
862
+ name: string;
863
+ description?: string;
864
+ createdAt: string;
865
+ updatedAt: string;
866
+ currentVersion: number;
867
+ versions: BenchmarkVersion[];
868
+ testCaseIds: string[];
869
+ runs: BenchmarkRun[];
870
+ }
871
+ export interface BenchmarkProgress {
872
+ currentTestCaseIndex: number;
873
+ startedCount?: number;
874
+ completedCount?: number;
875
+ totalTestCases: number;
876
+ currentRunId: string;
877
+ currentTestCaseId: string;
878
+ status: 'running' | 'completed' | 'failed' | 'cancelled';
879
+ }
880
+ export interface BenchmarkStartedEvent {
881
+ runId: string;
882
+ testCases: Array<{
883
+ id: string;
884
+ name: string;
885
+ status: 'pending';
886
+ }>;
887
+ }
888
+ /** @deprecated Use BenchmarkRunStatus instead */
889
+ export type ExperimentRunStatus = BenchmarkRunStatus;
890
+ /** @deprecated Use BenchmarkRun instead */
891
+ export type ExperimentRun = BenchmarkRun;
892
+ /** @deprecated Use Benchmark instead */
893
+ export type Experiment = Benchmark;
894
+ /** @deprecated Use BenchmarkProgress instead */
895
+ export type ExperimentProgress = BenchmarkProgress;
896
+ /** @deprecated Use BenchmarkStartedEvent instead */
897
+ export type ExperimentStartedEvent = BenchmarkStartedEvent;
898
+ /**
899
+ * Discriminator for documents in evals_benchmarks index.
900
+ * Legacy docs without this field default to 'benchmark' via normalization.
901
+ */
902
+ export type EvalDocType = 'benchmark' | 'evaluation-run' | 'benchmark-image';
903
+ /**
904
+ * BenchmarkImage — content-addressed snapshot of evaluation conditions
905
+ * ("the controls"): test-case contents + evaluator/judge conditions. Runs
906
+ * sharing a digest are comparable by construction; the digest is also the
907
+ * inherent dedup key (same command → same digest → same image, never a
908
+ * duplicate). Tags are docker-style mutable labels — never identity.
909
+ * Stored in the evals_benchmarks index/dir with docType 'benchmark-image'.
910
+ */
911
+ export interface BenchmarkImage {
912
+ id: string;
913
+ docType: 'benchmark-image';
914
+ digest: string;
915
+ tags: string[];
916
+ testCaseFingerprints: Array<{
917
+ id?: string;
918
+ name: string;
919
+ contentHash: string;
920
+ }>;
921
+ testCaseCount: number;
922
+ evalConditions: {
923
+ evaluatorId?: string;
924
+ judgeModelId?: string;
925
+ };
926
+ createdAt: string;
927
+ lastRunAt?: string;
928
+ }
929
+ /**
930
+ * Describes where test cases came from for an evaluation run.
931
+ * Multiple sources can be combined (union, deduplicated by test case ID).
932
+ */
933
+ export type TestCaseSource = {
934
+ type: 'benchmark';
935
+ benchmarkId: string;
936
+ benchmarkVersion?: number;
937
+ } | {
938
+ type: 'test-case-ids';
939
+ ids: string[];
940
+ } | {
941
+ type: 'file-import';
942
+ filenames: string[];
943
+ testCaseIds: string[];
944
+ } | {
945
+ type: 'code-import';
946
+ filenames: string[];
947
+ testCaseIds: string[];
948
+ } | {
949
+ type: 'directory-import';
950
+ dirPaths: string[];
951
+ testCaseIds: string[];
952
+ } | {
953
+ type: 'label-filter';
954
+ labels: string[];
955
+ };
956
+ /**
957
+ * EvaluationRun — first-class execution record.
958
+ * Stored as top-level doc in evals_benchmarks index with docType: 'evaluation-run'.
959
+ * Replaces embedded BenchmarkRun as primary execution entity.
960
+ */
961
+ export interface EvaluationRun {
962
+ id: string;
963
+ docType: 'evaluation-run';
964
+ name: string;
965
+ description?: string;
966
+ createdAt: string;
967
+ completedAt?: string;
968
+ status: BenchmarkRunStatus;
969
+ error?: string;
970
+ agentKey: string;
971
+ agentEndpoint?: string;
972
+ modelId: string;
973
+ /**
974
+ * Optional judge model id, distinct from {@link modelId} (the agent's
975
+ * LLM). Same precedence rules as on {@link BenchmarkRun.judgeModelId}.
976
+ */
977
+ judgeModelId?: string;
978
+ evaluatorId?: string;
979
+ headers?: Record<string, string>;
980
+ concurrency?: number;
981
+ sources: TestCaseSource[];
982
+ trigger: 'ui' | 'cli' | 'api' | 'schedule';
983
+ testCaseSnapshots: TestCaseSnapshot[];
984
+ results: Record<string, {
985
+ reportId: string;
986
+ status: RunResultStatus;
987
+ error?: string;
988
+ performanceMetrics?: TestCasePerformanceMetrics;
989
+ }>;
990
+ stats?: RunStats;
991
+ performanceMetrics?: RunPerformanceMetrics;
992
+ benchmarkId?: string;
993
+ benchmarkVersion?: number;
994
+ /**
995
+ * Content digest of this run's evaluation conditions (test-case contents +
996
+ * evaluator/judge conditions). Runs with equal digests ran under identical
997
+ * conditions and are directly comparable. See {@link BenchmarkImage}.
998
+ */
999
+ imageDigest?: string;
1000
+ /**
1001
+ * Provenance: id of the source {@link EvaluationRun} this run was created
1002
+ * from via `POST /api/storage/evaluation-runs/:id/rerun` ("kick off a
1003
+ * duplicate of the same run"). Undefined for runs created any other way.
1004
+ * The source run is NOT required to still exist for this run to be valid —
1005
+ * it's a point-in-time provenance link, not a live reference.
1006
+ */
1007
+ rerunOf?: string;
1008
+ }
1009
+ export interface TestCaseVersionRef {
1010
+ id: string;
1011
+ version: string;
1012
+ hash: string;
1013
+ }
1014
+ export interface RunAggregateMetrics {
1015
+ runId: string;
1016
+ runName: string;
1017
+ createdAt: string;
1018
+ modelId: string;
1019
+ agentKey: string;
1020
+ totalTestCases: number;
1021
+ passedCount: number;
1022
+ failedCount: number;
1023
+ /** Test cases the evaluator couldn't verdict (#242); excluded from pass rate. */
1024
+ erroredCount?: number;
1025
+ avgAccuracy: number;
1026
+ passRatePercent: number;
1027
+ totalTokens?: number;
1028
+ totalInputTokens?: number;
1029
+ totalOutputTokens?: number;
1030
+ totalCostUsd?: number;
1031
+ avgDurationMs?: number;
1032
+ totalLlmCalls?: number;
1033
+ totalToolCalls?: number;
1034
+ }
1035
+ export interface TestCaseRunResult {
1036
+ reportId?: string;
1037
+ status: 'completed' | 'failed' | 'missing';
1038
+ passFailStatus?: PassFailStatus;
1039
+ /**
1040
+ * Issue #242: when the evaluator could not produce a verdict
1041
+ * (`metricsStatus: 'error'` on the report), the comparison row carries
1042
+ * this flag so MetricCell can render an amber `Errored` chip distinct
1043
+ * from `Failed`. The legacy `passFailStatus` field on these reports is
1044
+ * cleared (`null`), so without this flag the cell would silently fall
1045
+ * through to `Failed` styling.
1046
+ */
1047
+ errored?: boolean;
1048
+ accuracy?: number;
1049
+ faithfulness?: number;
1050
+ trajectoryAlignment?: number;
1051
+ latencyScore?: number;
1052
+ testCaseVersion?: string;
1053
+ /** Error message if status is 'failed' */
1054
+ error?: string;
1055
+ }
1056
+ export interface TestCaseComparisonRow {
1057
+ testCaseId: string;
1058
+ testCaseName: string;
1059
+ labels: string[];
1060
+ /** @deprecated Use labels instead */
1061
+ category: Category;
1062
+ /** @deprecated Use labels instead */
1063
+ difficulty: Difficulty;
1064
+ results: Record<string, TestCaseRunResult>;
1065
+ hasVersionDifference: boolean;
1066
+ versions: string[];
1067
+ }
1068
+ export type RunConfigInput = Pick<BenchmarkRun, 'name' | 'description' | 'agentKey' | 'modelId' | 'judgeModelId' | 'agentEndpoint' | 'headers' | 'concurrency' | 'evaluatorId'>;
1069
+ import type { Request, Response } from 'express';
1070
+ export interface TypedRequest<T = any> extends Request {
1071
+ body: T;
1072
+ }
1073
+ export interface TypedResponse<T = any> extends Response {
1074
+ json: (body: T) => this;
1075
+ }
1076
+ export interface ExpectedStep {
1077
+ description: string;
1078
+ requiredTools?: string[];
1079
+ }
1080
+ export interface JudgeRequest {
1081
+ trajectory: TrajectoryStep[];
1082
+ expectedTrajectory: ExpectedStep[];
1083
+ expectedOutcomes?: string[];
1084
+ logs?: OpenSearchLog[];
1085
+ modelId?: string;
1086
+ evaluatorId?: string;
1087
+ }
1088
+ export interface JudgeResponse {
1089
+ passFailStatus: PassFailStatus;
1090
+ metrics: EvaluationMetrics;
1091
+ llmJudgeReasoning: string;
1092
+ improvementStrategies: ImprovementStrategy[];
1093
+ duration: number;
1094
+ /**
1095
+ * Set only by the demo/mock judge to flag that this verdict was NOT produced
1096
+ * by a real LLM (semi-random pass). A real provider never sets this. The
1097
+ * CLI/UI surface it so a "100% pass" from the mock can't be mistaken for a
1098
+ * real score.
1099
+ */
1100
+ warning?: string;
1101
+ }
1102
+ export interface AgentProxyRequest {
1103
+ endpoint: string;
1104
+ payload: any;
1105
+ headers?: Record<string, string>;
1106
+ }
1107
+ export interface StorageConfig {
1108
+ endpoint?: string;
1109
+ username?: string;
1110
+ password?: string;
1111
+ indexes: {
1112
+ testCases: string;
1113
+ benchmarks: string;
1114
+ runs: string;
1115
+ analytics: string;
1116
+ evaluators: string;
1117
+ };
1118
+ }
1119
+ export interface OpenSearchConfig {
1120
+ endpoint: string;
1121
+ username: string;
1122
+ password: string;
1123
+ indexPattern: string;
1124
+ }
1125
+ export interface LogsQuery {
1126
+ runId?: string;
1127
+ query?: string;
1128
+ startTime?: number;
1129
+ endTime?: number;
1130
+ size?: number;
1131
+ }
1132
+ export interface LogsResponse {
1133
+ hits: {
1134
+ hits: any[];
1135
+ total: any;
1136
+ };
1137
+ logs: OpenSearchLog[];
1138
+ total: number;
1139
+ }
1140
+ export interface HealthStatus {
1141
+ status: 'ok' | 'error' | 'not_configured';
1142
+ error?: string;
1143
+ errorCategory?: 'auth' | 'connection' | 'index_not_found' | 'unknown';
1144
+ suggestion?: string;
1145
+ index?: string;
1146
+ cluster?: any;
1147
+ }
1148
+ export interface AggregateMetrics {
1149
+ totalRuns: number;
1150
+ successRate: number;
1151
+ totalCostUsd: number;
1152
+ avgCostUsd: number;
1153
+ avgDurationMs: number;
1154
+ p50DurationMs: number;
1155
+ p95DurationMs: number;
1156
+ avgTokens: number;
1157
+ totalInputTokens: number;
1158
+ totalOutputTokens: number;
1159
+ avgLlmCalls: number;
1160
+ avgToolCalls: number;
1161
+ }
1162
+ export interface MetricsResult {
1163
+ runId: string;
1164
+ traceId: string | null;
1165
+ inputTokens: number;
1166
+ outputTokens: number;
1167
+ totalTokens: number;
1168
+ costUsd: number;
1169
+ durationMs: number;
1170
+ llmCalls: number;
1171
+ toolCalls: number;
1172
+ toolsUsed: string[];
1173
+ status: 'pending' | 'success' | 'error';
1174
+ }
1175
+ /**
1176
+ * Authentication type for OpenSearch clusters
1177
+ * - 'none': No authentication (e.g. local development clusters)
1178
+ * - 'basic': Username/password authentication (default, backwards compatible)
1179
+ * - 'sigv4': AWS SigV4 request signing for managed OpenSearch / Serverless
1180
+ */
1181
+ export type ClusterAuthType = 'none' | 'basic' | 'sigv4';
1182
+ /**
1183
+ * Base cluster configuration (endpoint + credentials)
1184
+ * Used for connecting to OpenSearch or other data sources
1185
+ */
1186
+ export interface ClusterConfig {
1187
+ endpoint: string;
1188
+ authType?: ClusterAuthType;
1189
+ username?: string;
1190
+ password?: string;
1191
+ awsProfile?: string;
1192
+ awsRegion?: string;
1193
+ awsService?: 'es' | 'aoss';
1194
+ tlsSkipVerify?: boolean;
1195
+ }
1196
+ /**
1197
+ * Storage cluster configuration - endpoint + credentials only
1198
+ * Index names (evals_test_cases, evals_experiments, evals_runs, evals_analytics)
1199
+ * are hardcoded in the adapter and not user-configurable.
1200
+ */
1201
+ export type StorageClusterConfig = ClusterConfig;
1202
+ /**
1203
+ * Observability cluster configuration - endpoint + credentials + OTEL index patterns
1204
+ * Used for traces, logs, and metrics from OpenTelemetry instrumentation
1205
+ */
1206
+ export interface ObservabilityClusterConfig extends ClusterConfig {
1207
+ indexes?: {
1208
+ traces?: string;
1209
+ logs?: string;
1210
+ metrics?: string;
1211
+ };
1212
+ }
1213
+ /**
1214
+ * Full data source configuration stored in localStorage
1215
+ * Both storage and observability can point to the same or different clusters
1216
+ */
1217
+ export interface DataSourceConfig {
1218
+ storage?: StorageClusterConfig;
1219
+ observability?: ObservabilityClusterConfig;
1220
+ }
1221
+ /**
1222
+ * Adapter type for data sources
1223
+ * 'file' is the default (JSON files in .agent-health/data/)
1224
+ * 'opensearch' when storage cluster is configured
1225
+ * 'memory' is for testing/demo
1226
+ */
1227
+ export type DataSourceAdapterType = 'file' | 'opensearch' | 'memory';
1228
+ export * from './skills.js';
1229
+ //# sourceMappingURL=index.d.ts.map