@opensearch-project/agent-health 0.3.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (518) hide show
  1. package/README.md +77 -6
  2. package/cli/dist/index.js +10072 -4502
  3. package/deployment/cloudformation/agent-health-observability.yaml +762 -0
  4. package/dist/assets/index-CCQRDlO0.js +243 -0
  5. package/dist/assets/index-CNHQVbcj.css +1 -0
  6. package/dist/index.html +2 -2
  7. package/docs/ARCHITECTURE.md +450 -0
  8. package/docs/BACKEND_JOB_QUEUE.md +405 -0
  9. package/docs/CLAUDE_CODE_TELEMETRY.md +283 -0
  10. package/docs/CLI.md +431 -0
  11. package/docs/CODING_AGENT_ANALYTICS.md +298 -0
  12. package/docs/CONFIGURATION.md +388 -0
  13. package/docs/CONNECTORS.md +536 -0
  14. package/docs/INSTRUMENT_WITH_OTEL.md +390 -0
  15. package/docs/ML-COMMONS-SETUP.md +289 -0
  16. package/docs/NPX_PACKAGING.md +195 -0
  17. package/docs/PERFORMANCE-MONITORING.md +200 -0
  18. package/docs/PERFORMANCE.md +390 -0
  19. package/docs/PI_PROFILING.md +169 -0
  20. package/docs/PLAN-non-agui-agent-support.md +525 -0
  21. package/docs/SDK.md +577 -0
  22. package/docs/SKILLS.md +264 -0
  23. package/docs/blogs/2026-02-28-opensearch-agent-health.md +200 -0
  24. package/docs/blogs/getting-started-blog.md +608 -0
  25. package/docs/diagrams/Agent-health.excalidraw +5656 -0
  26. package/docs/diagrams/architecture.png +0 -0
  27. package/docs/plans/field-redesign.md +468 -0
  28. package/docs/rfcs/001-coding-agent-analytics.md +374 -0
  29. package/docs/rfcs/002-enterprise-leaderboard.md +267 -0
  30. package/docs/rfcs/003-remote-aggregation.md +146 -0
  31. package/docs/rfcs/004-test-sdk-v2.md +599 -0
  32. package/docs/skills/AGENT_HEALTH.md +598 -0
  33. package/docs/skills/AGENT_PROFILE.md +191 -0
  34. package/docs/skills/add-connector/SKILL.md +68 -0
  35. package/docs/skills/agent-health-profile/SKILL.md +40 -0
  36. package/docs/skills/config-auth/SKILL.md +194 -0
  37. package/docs/skills/config-auth/evals/evals.json +35 -0
  38. package/docs/skills/create-pr/SKILL.md +73 -0
  39. package/docs/skills/instrument-otel/SKILL.md +84 -0
  40. package/docs/skills/write-test/SKILL.md +124 -0
  41. package/docs/ui prd.md +376 -0
  42. package/examples/README.md +53 -0
  43. package/examples/config/agent-health.config.example.ts +155 -0
  44. package/examples/connectors/echo-connector.ts +131 -0
  45. package/examples/eval-files/demo.eval.js +128 -0
  46. package/examples/eval-files/sdk-hooks-demo.eval.js +99 -0
  47. package/examples/pi-profiling/README.md +77 -0
  48. package/examples/pi-profiling/agent-health-profile.ts +417 -0
  49. package/lib/dist/lib/agentUtils.d.ts +29 -0
  50. package/lib/dist/lib/agentUtils.d.ts.map +1 -0
  51. package/lib/dist/lib/agentUtils.js +43 -0
  52. package/lib/dist/lib/agentUtils.js.map +1 -0
  53. package/lib/dist/lib/benchmarkExport.d.ts +14 -0
  54. package/lib/dist/lib/benchmarkExport.d.ts.map +1 -0
  55. package/lib/dist/lib/benchmarkExport.js +41 -0
  56. package/lib/dist/lib/benchmarkExport.js.map +1 -0
  57. package/lib/dist/lib/benchmarkVersionUtils.d.ts +37 -0
  58. package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -0
  59. package/lib/dist/lib/benchmarkVersionUtils.js +68 -0
  60. package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -0
  61. package/lib/dist/lib/config/defineConfig.d.ts +27 -0
  62. package/lib/dist/lib/config/defineConfig.d.ts.map +1 -0
  63. package/lib/dist/lib/config/defineConfig.js +28 -0
  64. package/lib/dist/lib/config/defineConfig.js.map +1 -0
  65. package/lib/dist/lib/config/index.d.ts +9 -0
  66. package/lib/dist/lib/config/index.d.ts.map +1 -0
  67. package/lib/dist/lib/config/index.js +8 -0
  68. package/lib/dist/lib/config/index.js.map +1 -0
  69. package/lib/dist/lib/config/loader.d.ts +39 -0
  70. package/lib/dist/lib/config/loader.d.ts.map +1 -0
  71. package/lib/dist/lib/config/loader.js +258 -0
  72. package/lib/dist/lib/config/loader.js.map +1 -0
  73. package/lib/dist/lib/config/statePaths.d.ts +61 -0
  74. package/lib/dist/lib/config/statePaths.d.ts.map +1 -0
  75. package/lib/dist/lib/config/statePaths.js +188 -0
  76. package/lib/dist/lib/config/statePaths.js.map +1 -0
  77. package/lib/dist/lib/config/types.d.ts +231 -0
  78. package/lib/dist/lib/config/types.d.ts.map +1 -0
  79. package/lib/dist/lib/config/types.js +6 -0
  80. package/lib/dist/lib/config/types.js.map +1 -0
  81. package/lib/dist/lib/config.d.ts +39 -0
  82. package/lib/dist/lib/config.d.ts.map +1 -0
  83. package/lib/dist/lib/config.js +118 -0
  84. package/lib/dist/lib/config.js.map +1 -0
  85. package/lib/dist/lib/constants.d.ts +70 -0
  86. package/lib/dist/lib/constants.d.ts.map +1 -0
  87. package/lib/dist/lib/constants.js +365 -0
  88. package/lib/dist/lib/constants.js.map +1 -0
  89. package/lib/dist/lib/contextUtilization.d.ts +23 -0
  90. package/lib/dist/lib/contextUtilization.d.ts.map +1 -0
  91. package/lib/dist/lib/contextUtilization.js +72 -0
  92. package/lib/dist/lib/contextUtilization.js.map +1 -0
  93. package/lib/dist/lib/dashboardMetrics.d.ts +87 -0
  94. package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -0
  95. package/lib/dist/lib/dashboardMetrics.js +242 -0
  96. package/lib/dist/lib/dashboardMetrics.js.map +1 -0
  97. package/lib/dist/lib/dataSourceConfig.d.ts +108 -0
  98. package/lib/dist/lib/dataSourceConfig.d.ts.map +1 -0
  99. package/lib/dist/lib/dataSourceConfig.js +166 -0
  100. package/lib/dist/lib/dataSourceConfig.js.map +1 -0
  101. package/lib/dist/lib/debug.d.ts +26 -0
  102. package/lib/dist/lib/debug.d.ts.map +1 -0
  103. package/lib/dist/lib/debug.js +132 -0
  104. package/lib/dist/lib/debug.js.map +1 -0
  105. package/lib/dist/lib/diagnostics.d.ts +28 -0
  106. package/lib/dist/lib/diagnostics.d.ts.map +1 -0
  107. package/lib/dist/lib/diagnostics.js +65 -0
  108. package/lib/dist/lib/diagnostics.js.map +1 -0
  109. package/lib/dist/lib/envCompat.d.ts +27 -0
  110. package/lib/dist/lib/envCompat.d.ts.map +1 -0
  111. package/lib/dist/lib/envCompat.js +73 -0
  112. package/lib/dist/lib/envCompat.js.map +1 -0
  113. package/lib/dist/lib/findPackageRoot.d.ts +7 -0
  114. package/lib/dist/lib/findPackageRoot.d.ts.map +1 -0
  115. package/lib/dist/lib/findPackageRoot.js +57 -0
  116. package/lib/dist/lib/findPackageRoot.js.map +1 -0
  117. package/lib/dist/lib/hooks.d.ts +36 -0
  118. package/lib/dist/lib/hooks.d.ts.map +1 -0
  119. package/lib/dist/lib/hooks.js +112 -0
  120. package/lib/dist/lib/hooks.js.map +1 -0
  121. package/lib/dist/lib/index.d.ts +47 -0
  122. package/lib/dist/lib/index.d.ts.map +1 -0
  123. package/lib/dist/lib/index.js +62 -0
  124. package/lib/dist/lib/index.js.map +1 -0
  125. package/lib/dist/lib/labels.d.ts +90 -0
  126. package/lib/dist/lib/labels.d.ts.map +1 -0
  127. package/lib/dist/lib/labels.js +158 -0
  128. package/lib/dist/lib/labels.js.map +1 -0
  129. package/lib/dist/lib/markdown.d.ts +16 -0
  130. package/lib/dist/lib/markdown.d.ts.map +1 -0
  131. package/lib/dist/lib/markdown.js +42 -0
  132. package/lib/dist/lib/markdown.js.map +1 -0
  133. package/lib/dist/lib/matchers/expect.d.ts +3 -0
  134. package/lib/dist/lib/matchers/expect.d.ts.map +1 -0
  135. package/lib/dist/lib/matchers/expect.js +225 -0
  136. package/lib/dist/lib/matchers/expect.js.map +1 -0
  137. package/lib/dist/lib/matchers/index.d.ts +8 -0
  138. package/lib/dist/lib/matchers/index.d.ts.map +1 -0
  139. package/lib/dist/lib/matchers/index.js +9 -0
  140. package/lib/dist/lib/matchers/index.js.map +1 -0
  141. package/lib/dist/lib/matchers/judgeAccessor.d.ts +113 -0
  142. package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -0
  143. package/lib/dist/lib/matchers/judgeAccessor.js +183 -0
  144. package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -0
  145. package/lib/dist/lib/matchers/session.d.ts +39 -0
  146. package/lib/dist/lib/matchers/session.d.ts.map +1 -0
  147. package/lib/dist/lib/matchers/session.js +116 -0
  148. package/lib/dist/lib/matchers/session.js.map +1 -0
  149. package/lib/dist/lib/matchers/traces.d.ts +55 -0
  150. package/lib/dist/lib/matchers/traces.d.ts.map +1 -0
  151. package/lib/dist/lib/matchers/traces.js +116 -0
  152. package/lib/dist/lib/matchers/traces.js.map +1 -0
  153. package/lib/dist/lib/matchers/types.d.ts +75 -0
  154. package/lib/dist/lib/matchers/types.d.ts.map +1 -0
  155. package/lib/dist/lib/matchers/types.js +6 -0
  156. package/lib/dist/lib/matchers/types.js.map +1 -0
  157. package/lib/dist/lib/packagePaths.d.ts +29 -0
  158. package/lib/dist/lib/packagePaths.d.ts.map +1 -0
  159. package/lib/dist/lib/packagePaths.js +63 -0
  160. package/lib/dist/lib/packagePaths.js.map +1 -0
  161. package/lib/dist/lib/performance.d.ts +51 -0
  162. package/lib/dist/lib/performance.d.ts.map +1 -0
  163. package/lib/dist/lib/performance.js +159 -0
  164. package/lib/dist/lib/performance.js.map +1 -0
  165. package/lib/dist/lib/portConfig.d.ts +29 -0
  166. package/lib/dist/lib/portConfig.d.ts.map +1 -0
  167. package/lib/dist/lib/portConfig.js +64 -0
  168. package/lib/dist/lib/portConfig.js.map +1 -0
  169. package/lib/dist/lib/preferences.d.ts +63 -0
  170. package/lib/dist/lib/preferences.d.ts.map +1 -0
  171. package/lib/dist/lib/preferences.js +117 -0
  172. package/lib/dist/lib/preferences.js.map +1 -0
  173. package/lib/dist/lib/resolveAgentModel.d.ts +22 -0
  174. package/lib/dist/lib/resolveAgentModel.d.ts.map +1 -0
  175. package/lib/dist/lib/resolveAgentModel.js +37 -0
  176. package/lib/dist/lib/resolveAgentModel.js.map +1 -0
  177. package/lib/dist/lib/runStats.d.ts +92 -0
  178. package/lib/dist/lib/runStats.d.ts.map +1 -0
  179. package/lib/dist/lib/runStats.js +160 -0
  180. package/lib/dist/lib/runStats.js.map +1 -0
  181. package/lib/dist/lib/telemetry/constants.d.ts +60 -0
  182. package/lib/dist/lib/telemetry/constants.d.ts.map +1 -0
  183. package/lib/dist/lib/telemetry/constants.js +87 -0
  184. package/lib/dist/lib/telemetry/constants.js.map +1 -0
  185. package/lib/dist/lib/telemetry/evalSpans.d.ts +61 -0
  186. package/lib/dist/lib/telemetry/evalSpans.d.ts.map +1 -0
  187. package/lib/dist/lib/telemetry/evalSpans.js +254 -0
  188. package/lib/dist/lib/telemetry/evalSpans.js.map +1 -0
  189. package/lib/dist/lib/telemetry/index.d.ts +11 -0
  190. package/lib/dist/lib/telemetry/index.d.ts.map +1 -0
  191. package/lib/dist/lib/telemetry/index.js +15 -0
  192. package/lib/dist/lib/telemetry/index.js.map +1 -0
  193. package/lib/dist/lib/telemetry/opensearchExporter.d.ts +43 -0
  194. package/lib/dist/lib/telemetry/opensearchExporter.d.ts.map +1 -0
  195. package/lib/dist/lib/telemetry/opensearchExporter.js +217 -0
  196. package/lib/dist/lib/telemetry/opensearchExporter.js.map +1 -0
  197. package/lib/dist/lib/telemetry/provider.d.ts +55 -0
  198. package/lib/dist/lib/telemetry/provider.d.ts.map +1 -0
  199. package/lib/dist/lib/telemetry/provider.js +140 -0
  200. package/lib/dist/lib/telemetry/provider.js.map +1 -0
  201. package/lib/dist/lib/testCaseLabels.d.ts +34 -0
  202. package/lib/dist/lib/testCaseLabels.d.ts.map +1 -0
  203. package/lib/dist/lib/testCaseLabels.js +88 -0
  204. package/lib/dist/lib/testCaseLabels.js.map +1 -0
  205. package/lib/dist/lib/testCaseValidation.d.ts +140 -0
  206. package/lib/dist/lib/testCaseValidation.d.ts.map +1 -0
  207. package/lib/dist/lib/testCaseValidation.js +162 -0
  208. package/lib/dist/lib/testCaseValidation.js.map +1 -0
  209. package/lib/dist/lib/testCases/agentFixture.d.ts +80 -0
  210. package/lib/dist/lib/testCases/agentFixture.d.ts.map +1 -0
  211. package/lib/dist/lib/testCases/agentFixture.js +43 -0
  212. package/lib/dist/lib/testCases/agentFixture.js.map +1 -0
  213. package/lib/dist/lib/testCases/authoringSurface.d.ts +10 -0
  214. package/lib/dist/lib/testCases/authoringSurface.d.ts.map +1 -0
  215. package/lib/dist/lib/testCases/authoringSurface.js +54 -0
  216. package/lib/dist/lib/testCases/authoringSurface.js.map +1 -0
  217. package/lib/dist/lib/testCases/codemod.d.ts +13 -0
  218. package/lib/dist/lib/testCases/codemod.d.ts.map +1 -0
  219. package/lib/dist/lib/testCases/codemod.js +169 -0
  220. package/lib/dist/lib/testCases/codemod.js.map +1 -0
  221. package/lib/dist/lib/testCases/define.d.ts +114 -0
  222. package/lib/dist/lib/testCases/define.d.ts.map +1 -0
  223. package/lib/dist/lib/testCases/define.js +253 -0
  224. package/lib/dist/lib/testCases/define.js.map +1 -0
  225. package/lib/dist/lib/testCases/evaluators.d.ts +80 -0
  226. package/lib/dist/lib/testCases/evaluators.d.ts.map +1 -0
  227. package/lib/dist/lib/testCases/evaluators.js +105 -0
  228. package/lib/dist/lib/testCases/evaluators.js.map +1 -0
  229. package/lib/dist/lib/testCases/index.d.ts +14 -0
  230. package/lib/dist/lib/testCases/index.d.ts.map +1 -0
  231. package/lib/dist/lib/testCases/index.js +12 -0
  232. package/lib/dist/lib/testCases/index.js.map +1 -0
  233. package/lib/dist/lib/testCases/judge.d.ts +165 -0
  234. package/lib/dist/lib/testCases/judge.d.ts.map +1 -0
  235. package/lib/dist/lib/testCases/judge.js +359 -0
  236. package/lib/dist/lib/testCases/judge.js.map +1 -0
  237. package/lib/dist/lib/testCases/loader.d.ts +26 -0
  238. package/lib/dist/lib/testCases/loader.d.ts.map +1 -0
  239. package/lib/dist/lib/testCases/loader.js +149 -0
  240. package/lib/dist/lib/testCases/loader.js.map +1 -0
  241. package/lib/dist/lib/testCases/types.d.ts +242 -0
  242. package/lib/dist/lib/testCases/types.d.ts.map +1 -0
  243. package/lib/dist/lib/testCases/types.js +6 -0
  244. package/lib/dist/lib/testCases/types.js.map +1 -0
  245. package/lib/dist/lib/theme.d.ts +6 -0
  246. package/lib/dist/lib/theme.d.ts.map +1 -0
  247. package/lib/dist/lib/theme.js +36 -0
  248. package/lib/dist/lib/theme.js.map +1 -0
  249. package/lib/dist/lib/uiTelemetry.d.ts +7 -0
  250. package/lib/dist/lib/uiTelemetry.d.ts.map +1 -0
  251. package/lib/dist/lib/uiTelemetry.js +25 -0
  252. package/lib/dist/lib/uiTelemetry.js.map +1 -0
  253. package/lib/dist/lib/utils.d.ts +96 -0
  254. package/lib/dist/lib/utils.d.ts.map +1 -0
  255. package/lib/dist/lib/utils.js +232 -0
  256. package/lib/dist/lib/utils.js.map +1 -0
  257. package/lib/dist/lib/workflow/consolidate.d.ts +12 -0
  258. package/lib/dist/lib/workflow/consolidate.d.ts.map +1 -0
  259. package/lib/dist/lib/workflow/consolidate.js +33 -0
  260. package/lib/dist/lib/workflow/consolidate.js.map +1 -0
  261. package/lib/dist/lib/workflow/index.d.ts +13 -0
  262. package/lib/dist/lib/workflow/index.d.ts.map +1 -0
  263. package/lib/dist/lib/workflow/index.js +12 -0
  264. package/lib/dist/lib/workflow/index.js.map +1 -0
  265. package/lib/dist/lib/workflow/ledger.d.ts +30 -0
  266. package/lib/dist/lib/workflow/ledger.d.ts.map +1 -0
  267. package/lib/dist/lib/workflow/ledger.js +41 -0
  268. package/lib/dist/lib/workflow/ledger.js.map +1 -0
  269. package/lib/dist/lib/workflow/pool.d.ts +13 -0
  270. package/lib/dist/lib/workflow/pool.d.ts.map +1 -0
  271. package/lib/dist/lib/workflow/pool.js +44 -0
  272. package/lib/dist/lib/workflow/pool.js.map +1 -0
  273. package/lib/dist/lib/workflow/source.d.ts +22 -0
  274. package/lib/dist/lib/workflow/source.d.ts.map +1 -0
  275. package/lib/dist/lib/workflow/source.js +29 -0
  276. package/lib/dist/lib/workflow/source.js.map +1 -0
  277. package/lib/dist/lib/workflow/stepB.d.ts +71 -0
  278. package/lib/dist/lib/workflow/stepB.d.ts.map +1 -0
  279. package/lib/dist/lib/workflow/stepB.js +99 -0
  280. package/lib/dist/lib/workflow/stepB.js.map +1 -0
  281. package/lib/dist/lib/workflow/types.d.ts +86 -0
  282. package/lib/dist/lib/workflow/types.d.ts.map +1 -0
  283. package/lib/dist/lib/workflow/types.js +6 -0
  284. package/lib/dist/lib/workflow/types.js.map +1 -0
  285. package/lib/dist/lib/workflow/workflow.d.ts +119 -0
  286. package/lib/dist/lib/workflow/workflow.d.ts.map +1 -0
  287. package/lib/dist/lib/workflow/workflow.js +195 -0
  288. package/lib/dist/lib/workflow/workflow.js.map +1 -0
  289. package/lib/dist/services/agent/aguiConverter.d.ts +50 -0
  290. package/lib/dist/services/agent/aguiConverter.d.ts.map +1 -0
  291. package/lib/dist/services/agent/aguiConverter.js +449 -0
  292. package/lib/dist/services/agent/aguiConverter.js.map +1 -0
  293. package/lib/dist/services/agent/index.d.ts +10 -0
  294. package/lib/dist/services/agent/index.d.ts.map +1 -0
  295. package/lib/dist/services/agent/index.js +12 -0
  296. package/lib/dist/services/agent/index.js.map +1 -0
  297. package/lib/dist/services/agent/payloadBuilder.d.ts +33 -0
  298. package/lib/dist/services/agent/payloadBuilder.d.ts.map +1 -0
  299. package/lib/dist/services/agent/payloadBuilder.js +75 -0
  300. package/lib/dist/services/agent/payloadBuilder.js.map +1 -0
  301. package/lib/dist/services/agent/sseStream.d.ts +43 -0
  302. package/lib/dist/services/agent/sseStream.d.ts.map +1 -0
  303. package/lib/dist/services/agent/sseStream.js +223 -0
  304. package/lib/dist/services/agent/sseStream.js.map +1 -0
  305. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.d.ts +44 -0
  306. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.d.ts.map +1 -0
  307. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.js +95 -0
  308. package/lib/dist/services/connectors/agui/AGUIStreamingConnector.js.map +1 -0
  309. package/lib/dist/services/connectors/base/BaseConnector.d.ts +81 -0
  310. package/lib/dist/services/connectors/base/BaseConnector.d.ts.map +1 -0
  311. package/lib/dist/services/connectors/base/BaseConnector.js +170 -0
  312. package/lib/dist/services/connectors/base/BaseConnector.js.map +1 -0
  313. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +116 -0
  314. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -0
  315. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +403 -0
  316. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -0
  317. package/lib/dist/services/connectors/index.d.ts +13 -0
  318. package/lib/dist/services/connectors/index.d.ts.map +1 -0
  319. package/lib/dist/services/connectors/index.js +32 -0
  320. package/lib/dist/services/connectors/index.js.map +1 -0
  321. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +48 -0
  322. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -0
  323. package/lib/dist/services/connectors/kiro/KiroConnector.js +158 -0
  324. package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -0
  325. package/lib/dist/services/connectors/langgraph/LangGraphConnector.d.ts +36 -0
  326. package/lib/dist/services/connectors/langgraph/LangGraphConnector.d.ts.map +1 -0
  327. package/lib/dist/services/connectors/langgraph/LangGraphConnector.js +175 -0
  328. package/lib/dist/services/connectors/langgraph/LangGraphConnector.js.map +1 -0
  329. package/lib/dist/services/connectors/mock/MockConnector.d.ts +37 -0
  330. package/lib/dist/services/connectors/mock/MockConnector.d.ts.map +1 -0
  331. package/lib/dist/services/connectors/mock/MockConnector.js +120 -0
  332. package/lib/dist/services/connectors/mock/MockConnector.js.map +1 -0
  333. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.d.ts +42 -0
  334. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.d.ts.map +1 -0
  335. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.js +133 -0
  336. package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.js.map +1 -0
  337. package/lib/dist/services/connectors/pi/PiConnector.d.ts +87 -0
  338. package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -0
  339. package/lib/dist/services/connectors/pi/PiConnector.js +274 -0
  340. package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -0
  341. package/lib/dist/services/connectors/registry.d.ts +57 -0
  342. package/lib/dist/services/connectors/registry.d.ts.map +1 -0
  343. package/lib/dist/services/connectors/registry.js +106 -0
  344. package/lib/dist/services/connectors/registry.js.map +1 -0
  345. package/lib/dist/services/connectors/rest/RESTConnector.d.ts +38 -0
  346. package/lib/dist/services/connectors/rest/RESTConnector.d.ts.map +1 -0
  347. package/lib/dist/services/connectors/rest/RESTConnector.js +117 -0
  348. package/lib/dist/services/connectors/rest/RESTConnector.js.map +1 -0
  349. package/lib/dist/services/connectors/server.d.ts +13 -0
  350. package/lib/dist/services/connectors/server.d.ts.map +1 -0
  351. package/lib/dist/services/connectors/server.js +34 -0
  352. package/lib/dist/services/connectors/server.js.map +1 -0
  353. package/lib/dist/services/connectors/strands/StrandsConnector.d.ts +48 -0
  354. package/lib/dist/services/connectors/strands/StrandsConnector.d.ts.map +1 -0
  355. package/lib/dist/services/connectors/strands/StrandsConnector.js +221 -0
  356. package/lib/dist/services/connectors/strands/StrandsConnector.js.map +1 -0
  357. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +88 -0
  358. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -0
  359. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +418 -0
  360. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -0
  361. package/lib/dist/services/connectors/types.d.ts +213 -0
  362. package/lib/dist/services/connectors/types.d.ts.map +1 -0
  363. package/lib/dist/services/connectors/types.js +6 -0
  364. package/lib/dist/services/connectors/types.js.map +1 -0
  365. package/lib/dist/services/evaluation/bedrockJudge.d.ts +64 -0
  366. package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -0
  367. package/lib/dist/services/evaluation/bedrockJudge.js +167 -0
  368. package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -0
  369. package/lib/dist/services/evaluation/evaluatorError.d.ts +56 -0
  370. package/lib/dist/services/evaluation/evaluatorError.d.ts.map +1 -0
  371. package/lib/dist/services/evaluation/evaluatorError.js +56 -0
  372. package/lib/dist/services/evaluation/evaluatorError.js.map +1 -0
  373. package/lib/dist/services/evaluation/index.d.ts +106 -0
  374. package/lib/dist/services/evaluation/index.d.ts.map +1 -0
  375. package/lib/dist/services/evaluation/index.js +684 -0
  376. package/lib/dist/services/evaluation/index.js.map +1 -0
  377. package/lib/dist/services/evaluation/mockTrajectory.d.ts +3 -0
  378. package/lib/dist/services/evaluation/mockTrajectory.d.ts.map +1 -0
  379. package/lib/dist/services/evaluation/mockTrajectory.js +72 -0
  380. package/lib/dist/services/evaluation/mockTrajectory.js.map +1 -0
  381. package/lib/dist/services/opensearch/client.d.ts +26 -0
  382. package/lib/dist/services/opensearch/client.d.ts.map +1 -0
  383. package/lib/dist/services/opensearch/client.js +131 -0
  384. package/lib/dist/services/opensearch/client.js.map +1 -0
  385. package/lib/dist/services/opensearch/index.d.ts +16 -0
  386. package/lib/dist/services/opensearch/index.d.ts.map +1 -0
  387. package/lib/dist/services/opensearch/index.js +25 -0
  388. package/lib/dist/services/opensearch/index.js.map +1 -0
  389. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +123 -0
  390. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -0
  391. package/lib/dist/services/storage/asyncBenchmarkStorage.js +429 -0
  392. package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -0
  393. package/lib/dist/services/storage/asyncRunStorage.d.ts +127 -0
  394. package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -0
  395. package/lib/dist/services/storage/asyncRunStorage.js +448 -0
  396. package/lib/dist/services/storage/asyncRunStorage.js.map +1 -0
  397. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +156 -0
  398. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -0
  399. package/lib/dist/services/storage/asyncTestCaseStorage.js +285 -0
  400. package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -0
  401. package/lib/dist/services/storage/index.d.ts +17 -0
  402. package/lib/dist/services/storage/index.d.ts.map +1 -0
  403. package/lib/dist/services/storage/index.js +20 -0
  404. package/lib/dist/services/storage/index.js.map +1 -0
  405. package/lib/dist/services/storage/migration.d.ts +54 -0
  406. package/lib/dist/services/storage/migration.d.ts.map +1 -0
  407. package/lib/dist/services/storage/migration.js +296 -0
  408. package/lib/dist/services/storage/migration.js.map +1 -0
  409. package/lib/dist/services/storage/opensearchClient.d.ts +924 -0
  410. package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -0
  411. package/lib/dist/services/storage/opensearchClient.js +435 -0
  412. package/lib/dist/services/storage/opensearchClient.js.map +1 -0
  413. package/lib/dist/services/traces/browserRecovery.d.ts +26 -0
  414. package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -0
  415. package/lib/dist/services/traces/browserRecovery.js +81 -0
  416. package/lib/dist/services/traces/browserRecovery.js.map +1 -0
  417. package/lib/dist/services/traces/categoryStyles.d.ts +21 -0
  418. package/lib/dist/services/traces/categoryStyles.d.ts.map +1 -0
  419. package/lib/dist/services/traces/categoryStyles.js +56 -0
  420. package/lib/dist/services/traces/categoryStyles.js.map +1 -0
  421. package/lib/dist/services/traces/executionOrderTransform.d.ts +35 -0
  422. package/lib/dist/services/traces/executionOrderTransform.d.ts.map +1 -0
  423. package/lib/dist/services/traces/executionOrderTransform.js +313 -0
  424. package/lib/dist/services/traces/executionOrderTransform.js.map +1 -0
  425. package/lib/dist/services/traces/fetchSpansForRun.d.ts +86 -0
  426. package/lib/dist/services/traces/fetchSpansForRun.d.ts.map +1 -0
  427. package/lib/dist/services/traces/fetchSpansForRun.js +69 -0
  428. package/lib/dist/services/traces/fetchSpansForRun.js.map +1 -0
  429. package/lib/dist/services/traces/flowTransform.d.ts +24 -0
  430. package/lib/dist/services/traces/flowTransform.d.ts.map +1 -0
  431. package/lib/dist/services/traces/flowTransform.js +228 -0
  432. package/lib/dist/services/traces/flowTransform.js.map +1 -0
  433. package/lib/dist/services/traces/index.d.ts +121 -0
  434. package/lib/dist/services/traces/index.d.ts.map +1 -0
  435. package/lib/dist/services/traces/index.js +255 -0
  436. package/lib/dist/services/traces/index.js.map +1 -0
  437. package/lib/dist/services/traces/intentTransform.d.ts +20 -0
  438. package/lib/dist/services/traces/intentTransform.d.ts.map +1 -0
  439. package/lib/dist/services/traces/intentTransform.js +131 -0
  440. package/lib/dist/services/traces/intentTransform.js.map +1 -0
  441. package/lib/dist/services/traces/judgeAgentsHints.d.ts +63 -0
  442. package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -0
  443. package/lib/dist/services/traces/judgeAgentsHints.js +89 -0
  444. package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -0
  445. package/lib/dist/services/traces/messageExtraction.d.ts +15 -0
  446. package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -0
  447. package/lib/dist/services/traces/messageExtraction.js +251 -0
  448. package/lib/dist/services/traces/messageExtraction.js.map +1 -0
  449. package/lib/dist/services/traces/spanCategorization.d.ts +63 -0
  450. package/lib/dist/services/traces/spanCategorization.d.ts.map +1 -0
  451. package/lib/dist/services/traces/spanCategorization.js +276 -0
  452. package/lib/dist/services/traces/spanCategorization.js.map +1 -0
  453. package/lib/dist/services/traces/spanPreprocessing.d.ts +37 -0
  454. package/lib/dist/services/traces/spanPreprocessing.d.ts.map +1 -0
  455. package/lib/dist/services/traces/spanPreprocessing.js +102 -0
  456. package/lib/dist/services/traces/spanPreprocessing.js.map +1 -0
  457. package/lib/dist/services/traces/spansToTrajectory.d.ts +36 -0
  458. package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -0
  459. package/lib/dist/services/traces/spansToTrajectory.js +387 -0
  460. package/lib/dist/services/traces/spansToTrajectory.js.map +1 -0
  461. package/lib/dist/services/traces/toolSimilarity.d.ts +35 -0
  462. package/lib/dist/services/traces/toolSimilarity.d.ts.map +1 -0
  463. package/lib/dist/services/traces/toolSimilarity.js +203 -0
  464. package/lib/dist/services/traces/toolSimilarity.js.map +1 -0
  465. package/lib/dist/services/traces/traceComparison.d.ts +31 -0
  466. package/lib/dist/services/traces/traceComparison.d.ts.map +1 -0
  467. package/lib/dist/services/traces/traceComparison.js +318 -0
  468. package/lib/dist/services/traces/traceComparison.js.map +1 -0
  469. package/lib/dist/services/traces/traceGrouping.d.ts +19 -0
  470. package/lib/dist/services/traces/traceGrouping.d.ts.map +1 -0
  471. package/lib/dist/services/traces/traceGrouping.js +107 -0
  472. package/lib/dist/services/traces/traceGrouping.js.map +1 -0
  473. package/lib/dist/services/traces/tracePoller.d.ts +84 -0
  474. package/lib/dist/services/traces/tracePoller.d.ts.map +1 -0
  475. package/lib/dist/services/traces/tracePoller.js +309 -0
  476. package/lib/dist/services/traces/tracePoller.js.map +1 -0
  477. package/lib/dist/services/traces/traceStats.d.ts +45 -0
  478. package/lib/dist/services/traces/traceStats.d.ts.map +1 -0
  479. package/lib/dist/services/traces/traceStats.js +114 -0
  480. package/lib/dist/services/traces/traceStats.js.map +1 -0
  481. package/lib/dist/services/traces/traceSummary.d.ts +47 -0
  482. package/lib/dist/services/traces/traceSummary.d.ts.map +1 -0
  483. package/lib/dist/services/traces/traceSummary.js +68 -0
  484. package/lib/dist/services/traces/traceSummary.js.map +1 -0
  485. package/lib/dist/services/traces/utils.d.ts +33 -0
  486. package/lib/dist/services/traces/utils.d.ts.map +1 -0
  487. package/lib/dist/services/traces/utils.js +114 -0
  488. package/lib/dist/services/traces/utils.js.map +1 -0
  489. package/lib/dist/types/agui.d.ts +13 -0
  490. package/lib/dist/types/agui.d.ts.map +1 -0
  491. package/lib/dist/types/agui.js +16 -0
  492. package/lib/dist/types/agui.js.map +1 -0
  493. package/lib/dist/types/index.d.ts +1175 -0
  494. package/lib/dist/types/index.d.ts.map +1 -0
  495. package/lib/dist/types/index.js +12 -0
  496. package/lib/dist/types/index.js.map +1 -0
  497. package/lib/dist/types/skills.d.ts +146 -0
  498. package/lib/dist/types/skills.d.ts.map +1 -0
  499. package/lib/dist/types/skills.js +6 -0
  500. package/lib/dist/types/skills.js.map +1 -0
  501. package/observio-sample-agent/pi-package/README.md +112 -0
  502. package/observio-sample-agent/pi-package/extensions/agent-health.ts +373 -0
  503. package/observio-sample-agent/pi-package/package.json +17 -0
  504. package/observio-sample-agent/pi-package/prompts/agent-health.md +37 -0
  505. package/observio-sample-agent/pi-package/skills/create-pr/SKILL.md +88 -0
  506. package/observio-sample-agent/pi-package/skills/fix-bug/SKILL.md +71 -0
  507. package/observio-sample-agent/pi-package/skills/implement-feature/SKILL.md +156 -0
  508. package/observio-sample-agent/pi-package/skills/instrument-otel/SKILL.md +208 -0
  509. package/observio-sample-agent/pi-package/skills/setup-collector/SKILL.md +146 -0
  510. package/observio-sample-agent/pi-package/skills/write-test/SKILL.md +115 -0
  511. package/package.json +64 -13
  512. package/server/dist/app.js +32651 -17637
  513. package/server/dist/index.js +29875 -14638
  514. package/tsconfig.lib.json +71 -0
  515. package/dist/assets/index-EvPLSTAS.js +0 -267
  516. package/dist/assets/index-RXasQKUs.css +0 -1
  517. package/lib/dist/config/index.js +0 -404
  518. package/lib/dist/index.js +0 -1665
package/docs/SDK.md ADDED
@@ -0,0 +1,577 @@
1
+ <!--
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ -->
5
+
6
+ # Agent Health SDK Guide (experimental)
7
+
8
+ > ⚠️ **Experimental.** The SDK API surface — `test()` signature, options shape,
9
+ > fixtures, matcher set — may change in a minor release without a deprecation
10
+ > cycle. Pin your `@opensearch-project/agent-health` version if you depend on
11
+ > it. Set `AH_SUPPRESS_EXPERIMENTAL=1` to silence the runtime notice.
12
+
13
+ The SDK lets you write Agent Health test cases as plain JavaScript / TypeScript
14
+ files instead of clicking through the UI. Tests live alongside your repo,
15
+ follow Playwright's mental model (`test`, `expect`, fixtures), and produce
16
+ **per-matcher results** that the UI renders as a structured breakdown.
17
+
18
+ ```javascript
19
+ const { test, expect } = require('@opensearch-project/agent-health');
20
+
21
+ test('rca-log-analysis', {
22
+ prompt: 'Why is service X failing?',
23
+ labels: ['category:RCA', 'difficulty:Medium'],
24
+ }, async function ({ agent, judge, expect }) {
25
+ const result = await agent.run(); // invoke the agent (once)
26
+ expect(result.agentOutput).to.contain('root cause');
27
+ expect(result.trajectory).to.haveCalledTool('search_logs');
28
+ expect(result).to.haveCompletedWithin(60_000);
29
+ await judge(result, 'identifies the failing dependency'); // gate
30
+ expect(result.traces.totalTokens).to.be.lessThan(10_000);
31
+ });
32
+ ```
33
+
34
+ > **v2 control inversion (RFC 004).** The test body now drives the agent via
35
+ > `await agent.run()` — like Playwright's `await page.goto()`. The older eager
36
+ > form (a pre-populated `result` fixture) still works during the transition;
37
+ > see [Control inversion](#3-control-inversion--the-agent-fixture).
38
+
39
+ ---
40
+
41
+ ## Concepts
42
+
43
+ ### 1. `test()` — registers a test case
44
+
45
+ Two valid signatures:
46
+
47
+ ```javascript
48
+ test('name', body) // no options
49
+ test('name', options, body) // with options
50
+ ```
51
+
52
+ Only `name` is required. All `options` fields are optional. Within the same
53
+ `.eval.js` / `.eval.ts` file every test name must be unique — registration
54
+ throws on duplicates.
55
+
56
+ ### 2. The body receives **fixtures**
57
+
58
+ ```javascript
59
+ async function ({ agent, judge, evaluate, expect, testInfo, provisioned }) { ... }
60
+ ```
61
+
62
+ | Fixture | Type | What it gives you |
63
+ |----------|--------------------------------------------------|-------------------|
64
+ | `agent` | `AgentFixture` | `await agent.run(prompt?, options?)` — invoke the agent **once** (control inversion). Returns an `EvalResult`. |
65
+ | `judge` | `JudgeFn` | Non-throwing LLM judge. `judge(...)` gates; `judge.observe(...)` is observational. Returns a `Verdict`. |
66
+ | `evaluate` | `EvaluateFn` | Run a custom programmatic evaluator registered with `defineEvaluator()`. |
67
+ | `expect` | chai's `expect` with our recording plugin | Synchronous matcher entry-point |
68
+ | `testInfo` | `TestInfo` (read-only) | `{ name, benchmarkPath, sourceFile, testCaseId }` |
69
+ | `provisioned` | `Readonly<Record<string, unknown>>` | Values set by `beforeEach` via `provide()` |
70
+ | `result` | `EvalResult` *(legacy eager path)* | Pre-populated agent result — present when the runner invokes before the body. Prefer `await agent.run()`. |
71
+ | `traces` | `TracesAccessor` *(legacy)* | Standalone accessor; equivalent to `result.traces` after `agent.run()`. |
72
+
73
+ `expect` is also exported at the top level for convenience. Both are the same
74
+ function. `result.traces` exposes the same OTel accessor as the standalone
75
+ `traces` fixture, scoped to the run.
76
+
77
+ ### 3. Control inversion — the `agent` fixture
78
+
79
+ The test body invokes the agent itself with `await agent.run()`, the way a
80
+ Playwright test calls `await page.goto()` — so you can do setup *before* the
81
+ agent runs (seed a ticket, provision a workspace) and feed the result in:
82
+
83
+ ```javascript
84
+ test('payment RCA', { prompt: 'Triage the latest payment incident' },
85
+ async ({ agent, expect, judge }) => {
86
+ const id = await seedTicket(); // setup BEFORE the agent
87
+ const result = await agent.run(`Triage ${id}`); // per-call prompt wins over the option
88
+ expect(result.trajectory).to.haveCalledTool('search_logs');
89
+ await judge(result, 'identifies the DB outage');
90
+ });
91
+ ```
92
+
93
+ - **Exactly one invocation per test (enforced).** One test ⇒ one invocation ⇒
94
+ one comparable trajectory. A second `agent.run()` throws. Multi-turn
95
+ conversations (if a connector models them) happen inside that single run.
96
+ - **`agent.run(prompt?, options?)`** — `prompt` defaults to the test's `prompt`
97
+ option; `options` is `{ context?, env? }` (e.g. pass a provisioned workspace
98
+ dir via `env`). Returns a fully-captured `EvalResult` with `result.traces`.
99
+ - **`agent.invoked`** — read-only boolean, true once `run()` has been called.
100
+ - **Legacy eager path.** Older tests destructure a pre-populated `result`
101
+ fixture. That still works; new tests should prefer `agent.run()`. A codemod
102
+ converts the old shape — see [Migrating v1 → v2](#migrating-v1--v2).
103
+
104
+ ### 4. Lifecycle hooks (`beforeEach` / `afterEach` / `beforeAll` / `afterAll`)
105
+
106
+ For *side-effecting* per-test setup with a teardown step — the kind that
107
+ a connector can't express because connectors are pure request-shapers
108
+ with no lifecycle — the SDK provides Playwright-style hooks. Use them
109
+ for things like materializing a temp workspace, seeding a database,
110
+ starting a sandbox, or writing a fixture file the agent's tools open.
111
+
112
+ ```javascript
113
+ const fs = require('fs');
114
+ const os = require('os');
115
+ const path = require('path');
116
+ const { test, beforeAll, afterAll, beforeEach, afterEach, expect } = require('@opensearch-project/agent-health');
117
+
118
+ let suiteRoot;
119
+ beforeAll(() => {
120
+ suiteRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'my-suite-'));
121
+ });
122
+ afterAll(() => fs.rmSync(suiteRoot, { recursive: true, force: true }));
123
+
124
+ beforeEach(({ provide, testInfo }) => {
125
+ const dir = fs.mkdtempSync(path.join(suiteRoot, `${testInfo.name}-`));
126
+ provide('workspaceDir', dir);
127
+ });
128
+ afterEach(({ provisioned }) => {
129
+ // afterEach always runs — even when beforeEach failed before the
130
+ // provide() call — so guard before reaching for the value.
131
+ if (typeof provisioned.workspaceDir === 'string') {
132
+ fs.rmSync(provisioned.workspaceDir, { recursive: true, force: true });
133
+ }
134
+ });
135
+
136
+ test('uses workspace', { prompt: '...' }, async ({ result, provisioned }) => {
137
+ expect(fs.existsSync(provisioned.workspaceDir)).to.equal(true);
138
+ });
139
+ ```
140
+
141
+ Key points:
142
+
143
+ - **Scope.** Hooks attach to the file they're declared in (top level) or
144
+ to the surrounding `describe()` block. Nested describes inherit outer
145
+ hooks; ordering is outer-→inner for `beforeAll`/`beforeEach`, and
146
+ inner-→outer for `afterEach`/`afterAll`.
147
+ - **Once-per-scope guarantee.** `beforeAll` runs exactly once, even when
148
+ the runner dispatches tests in parallel — the orchestrator uses a
149
+ promise-based once-latch and all parallel arrivals await it. `afterAll`
150
+ uses a remaining-tests counter and fires when the last test in the
151
+ scope completes.
152
+ - **Always-runs teardown.** `afterEach` and `afterAll` run even when the
153
+ body or `beforeEach` threw. Hook errors don't crash the runner; they
154
+ surface as `MatcherResult` entries on the test (visible in the same
155
+ per-matcher panel as your assertion failures).
156
+ - **`provide(key, value)`** is the only way to surface a value from
157
+ `beforeEach` to the body. Don't mutate `testInfo` and don't rely on
158
+ closure variables for per-test state — those don't survive parallelism.
159
+ The `provisioned` bag is per-test (each test gets a fresh empty object),
160
+ so concurrent tests are isolated.
161
+ - **`testInfo`** is read-only metadata: `{ name, benchmarkPath, sourceFile,
162
+ testCaseId }`. Useful for naming temp resources or tagging logs.
163
+ - **`test.beforeEach(...)`** is also accepted as an equivalent alias for
164
+ `beforeEach(...)`, mirroring Playwright's surface.
165
+
166
+ Hooks are a no-op when no test in the run uses them — the orchestrator
167
+ is short-circuited to a noop variant and existing tests pay zero cost.
168
+
169
+ See the demo at [`evals/sdk-hooks-demo.eval.js`](../evals/sdk-hooks-demo.eval.js).
170
+
171
+ ### 5. Matchers record structured verdicts
172
+
173
+ Every `expect(...).to.X(...)` call, every `judge(result, ...)` call, and every
174
+ traces helper produces one **MatcherResult**. The runner collects them and
175
+ the UI shows a per-matcher breakdown:
176
+
177
+ ```
178
+ Matchers (4/5 passed)
179
+ ─────────────────────
180
+ ✅ to contain 'root cause' [code]
181
+ ✅ haveCalledTool('search_logs') [code]
182
+ ❌ to be lessThan 30000 [code] actual: 47320
183
+ ✅ identifies the failing dependency [judge] score: 85%
184
+ ✅ totalTokens < 10000 [traces] 2,341 tokens
185
+ ```
186
+
187
+ This is the major upgrade over throw-and-fail: every assertion gets its own
188
+ row, status, and detail block.
189
+
190
+ ---
191
+
192
+ ## Test options
193
+
194
+ ```typescript
195
+ interface TestOptions {
196
+ prompt?: string; // Initial prompt sent to the agent
197
+ description?: string; // Free-form description shown in the UI
198
+ context?: { description: string; value: string }[];
199
+ labels?: string[]; // Prefixed strings: 'category:RCA', 'difficulty:Medium'
200
+ timeout?: number; // Per-test timeout override in ms
201
+ }
202
+ ```
203
+
204
+ ### No `category` / `difficulty` keys
205
+
206
+ The previous standalone `category` and `difficulty` fields are gone. They
207
+ live in `labels` as prefixed strings:
208
+
209
+ ```javascript
210
+ labels: ['category:RCA', 'difficulty:Medium', 'team:platform', 'tier:p0']
211
+ ```
212
+
213
+ Anything before the colon is the facet; anything after is the value. Free-form
214
+ labels without a colon are also fine. The UI extracts category/difficulty for
215
+ display via `lib/testCaseLabels.ts`. A cold-start migration on the server
216
+ auto-folds legacy top-level fields into labels for older documents.
217
+
218
+ ### No prompt = no agent invocation
219
+
220
+ ```javascript
221
+ test('data-quality-check', {
222
+ description: 'Verify fixtures match a baseline',
223
+ labels: ['category:Data Quality'],
224
+ }, function ({ result }) {
225
+ // result is empty — no agent was invoked
226
+ const baseline = JSON.parse(fs.readFileSync('./baselines.json'));
227
+ expect(baseline.version).to.equal(1);
228
+ });
229
+ ```
230
+
231
+ When `prompt` is omitted the runner skips agent invocation entirely and the
232
+ body runs against an empty `EvalResult` (durationMs: 0, trajectory: []).
233
+ Useful for purely data-driven tests where there's no agent step.
234
+
235
+ ---
236
+
237
+ ## EvalResult shape
238
+
239
+ ```typescript
240
+ interface EvalResult {
241
+ trajectory: TrajectoryAccessor; // see below
242
+ agentOutput: string; // concatenated final response text
243
+ finalResponse(): string; // sugar — same as agentOutput
244
+ parsedOutput(): unknown; // try-parse agentOutput as JSON
245
+ rawEvents: any[]; // raw AG-UI events
246
+ runId?: string; // for log/trace correlation
247
+ durationMs: number; // wall-clock duration (0 when no prompt)
248
+ tokenUsage?: { prompt; completion; total };
249
+ }
250
+ ```
251
+
252
+ ### Trajectory sugar accessors
253
+
254
+ Beyond being a normal `TrajectoryStep[]`, the trajectory has three helper
255
+ methods you can use without writing filter loops:
256
+
257
+ ```javascript
258
+ result.trajectory.stepsOfType('action'); // → all action steps
259
+ result.trajectory.toolCalls(); // → all action steps (alias)
260
+ result.trajectory.toolCalls('search_logs'); // → filtered by tool name
261
+ result.trajectory.firstToolCall('http_probe', { method: 'POST' });
262
+ // → { ...step, index: N } or null
263
+ ```
264
+
265
+ `firstToolCall` returns the matched step with an `.index` annotation so you
266
+ can assert ordering: `expect(firstSearch.index).to.be.lessThan(firstReview.index)`.
267
+
268
+ ---
269
+
270
+ ## Matchers
271
+
272
+ ### Built-in chai matchers
273
+
274
+ Every chai BDD matcher works (`.equal`, `.contain`, `.have.length.greaterThan`,
275
+ `.match`, etc.). See https://www.chaijs.com/api/bdd/ for the full reference.
276
+
277
+ ### Custom matchers
278
+
279
+ | Matcher | What it does |
280
+ |---------------------------------------------|--------------|
281
+ | `expect(traj).to.haveCalledTool(name, args?)` | At least one `action` step matches; `args` is partial-superset |
282
+ | `expect(traj).to.haveStepsOfType(type)` | At least one step of given type exists |
283
+ | `expect(text).to.haveOutputMatching(re)` | String matches regex (or contains substring when given a string) |
284
+ | `expect(result).to.haveCompletedWithin(ms)` | `result.durationMs` ≤ threshold |
285
+
286
+ ### LLM judge — `judge()`
287
+
288
+ ```javascript
289
+ const v = await judge(result, 'identifies the root cause'); // gate role
290
+ await judge.observe(result, 'mentions the runbook'); // observe (non-gating)
291
+ await judge(result, 'proposes a remediation', { model: 'claude-sonnet' });
292
+ await judge(result, 'follows the SOP', { evaluatorId: 'system-rca-default' });
293
+ ```
294
+
295
+ Calls the server's `/api/judge` endpoint with the run's trajectory plus the
296
+ user-supplied claim as the expected outcome, and records a `MatcherResult`
297
+ with the judge's score and reasoning.
298
+
299
+ **Non-throwing (RFC 004).** `judge(...)` returns a `Verdict`
300
+ (`{ pass, passFailStatus, score, reasoning, role, skipped, orThrow() }`)
301
+ **without throwing**. A failing `gate`-role verdict still fails the test (the
302
+ runner inspects the recorded MatcherResult); `judge.observe(...)` feeds
303
+ score + insights only and never fails the run. Call `.orThrow()` on a verdict
304
+ for the old bail-on-first-failure behaviour at a specific point.
305
+
306
+ - **`judge(result, claim)`** — *gate*: a failing verdict fails the test.
307
+ - **`judge.observe(result, claim)`** — *observe*: records score/reasoning, never gates.
308
+ - **`{ skip: true }`** (or `AH_SKIP_JUDGE=1`) — returns a non-gating `skipped`
309
+ verdict with no HTTP call; `{ skip: false }` forces the judge to run even
310
+ when `AH_SKIP_JUDGE` is set.
311
+
312
+ The legacy form `judge(trajectory, [...claims])` is preserved for backward
313
+ compatibility with code written against the original PR.
314
+
315
+ #### Per-call options
316
+
317
+ | Option | Forwarded as | What it does |
318
+ |---------------|--------------|--------------|
319
+ | `model` | `modelId` | Override the judge model. Same provider routing (`bedrock`, `litellm`, `claude-code`, `pi`, `openai-compatible`, `agentic`, `demo`) the UI uses. |
320
+ | `evaluatorId` | `evaluatorId`| Pick a stored evaluator. Same shape the UI sends — built-in ids are prefixed `system-` (e.g. `system-rca-default`, `system-factuality`) and resolve via `getSystemEvaluatorById`; anything else is a storage id resolved via `storage.evaluators.getById`. |
321
+ | `serverUrl` | (request URL)| Point at a non-default agent-health server (defaults to `http://localhost:${AGENT_HEALTH_PORT ?? 4001}`). |
322
+ | `skip` | (no request) | Tri-state. `true` → skip the judge (records a non-gating `skipped` verdict, no HTTP call). `false` → force the judge to run **even if `AH_SKIP_JUDGE` is set**. Omitted → defer to `AH_SKIP_JUDGE`. |
323
+
324
+ #### Run-level evaluator (UI-equivalent)
325
+
326
+ When the runner constructs the `judge` fixture for a test body, it binds
327
+ the **run-level** `evaluatorId` and judge `model` from the `EvaluationRun`
328
+ so destructured `judge` calls inherit the run's evaluator without the
329
+ author passing it manually:
330
+
331
+ ```javascript
332
+ // In the test body — no per-call evaluatorId needed.
333
+ test('rca-investigate', { prompt: 'Investigate the failing service ...' }, async ({ agent, judge }) => {
334
+ const result = await agent.run();
335
+ // If the run was created with `evaluatorId: 'system-rca-default'`, this
336
+ // call POSTs `{ ..., evaluatorId: 'system-rca-default' }` automatically.
337
+ // Substitute any user-defined evaluator id and the same binding applies.
338
+ await judge(result, 'identifies the ticket details');
339
+ await judge(result, 'reports the current state');
340
+ await judge(result, 'recommends concrete next steps');
341
+ });
342
+ ```
343
+
344
+ This matches the UI "Run Test" path exactly: pick an evaluator on the run
345
+ config, every judged test case in the run uses it. Per-call options always
346
+ win over the bound default — useful when one matcher in a test needs a
347
+ different evaluator:
348
+
349
+ ```javascript
350
+ await judge(result, 'meets product gap criteria', { evaluatorId: 'product-gap-eval' });
351
+ ```
352
+
353
+ The **imported** `judge` (from `require('@opensearch-project/agent-health')`)
354
+ is always the unbound version — use it when you genuinely want the server's
355
+ default evaluator regardless of run config:
356
+
357
+ ```javascript
358
+ const { judge } = require('@opensearch-project/agent-health');
359
+ // Always uses the server default evaluator; bypasses any run-level binding.
360
+ await judge(result, 'baseline check');
361
+ ```
362
+
363
+ In practice, prefer the fixture-destructured form so SDK runs and UI runs
364
+ produce comparable verdicts.
365
+
366
+ ### Custom evaluator prompts and provider parity
367
+
368
+ The saved evaluator's `systemPrompt` and `scoringConfig.metrics` are honored
369
+ on **every** judge provider — `bedrock`, `openai-compatible`, `litellm`,
370
+ `claude-code`, `pi`, `agent` (trace), and `agentic`. When the judge model's
371
+ `provider` (resolved from the model config or `evaluator.inferenceConfig`)
372
+ changes, the same prompt and rubric flow through unchanged. Two
373
+ provider-specific addendums are appended automatically because they
374
+ document the provider's tool-use contract, not the rubric:
375
+
376
+ - `agentic`: `AGENTIC_JUDGE_ADDENDUM` (the model is told it can iterate /
377
+ use tools / cross-reference).
378
+ - `agent` (trace): a `query_spans` / `query_logs` paragraph so the judge
379
+ knows the run-scoped trace tools are available, even when the saved
380
+ prompt doesn't mention them.
381
+
382
+ The saved prompt fully replaces the default base in both cases — a
383
+ regression test pins this for the `agent` provider in
384
+ [tests/unit/server/services/piAgenticJudgeService.test.ts](../tests/unit/server/services/piAgenticJudgeService.test.ts).
385
+
386
+ ### Surfacing extra judge-emitted fields
387
+
388
+ If the saved system prompt asks the model for fields beyond the typed wire
389
+ shape (`pass_fail_status` / `reasoning` / `metrics` / `improvement_strategies`),
390
+ those extra keys are captured into
391
+ `LLMJudgeResponse.extraFields` and rendered on the run-detail page's "Judge
392
+ Debug" section. There's no code change required to surface a new field —
393
+ ask for it in the prompt and it shows up. Numeric metrics inside `metrics`
394
+ that aren't declared in `scoringConfig.metrics` are split into
395
+ `extraFields.metrics_unmapped` so nothing the judge emits is silently
396
+ dropped.
397
+
398
+ ### Inspecting what reached the model (`AH_JUDGE_DEBUG`)
399
+
400
+ When iterating on a judge prompt, set `AH_JUDGE_DEBUG=1` on the server (or
401
+ run in dev — it's the default in `NODE_ENV !== 'production'`) to capture
402
+ the **system prompt** and **user prompt** the judge actually saw, plus the
403
+ **raw response** before parsing. They land on
404
+ `LLMJudgeResponse.judgeDebug` and render under "Judge Debug" on the
405
+ run-detail page — use them to confirm in one round whether your prompt
406
+ edit reached the model. Disabled by default in prod because system prompts
407
+ can be 10–20 KB and shipping them on every run bloats persisted run docs.
408
+
409
+ ### Custom evaluators — `defineEvaluator()` / `evaluate()`
410
+
411
+ Not every check is an LLM judge or a chai assertion — sometimes ground truth
412
+ lives in code (a SQL result must match a golden row set, a JSON answer must
413
+ validate against a schema). Register a deterministic evaluator once and call it
414
+ by id:
415
+
416
+ ```javascript
417
+ const { defineEvaluator, test } = require('@opensearch-project/agent-health');
418
+
419
+ defineEvaluator('sql-matches-golden', ({ result }) => {
420
+ const rows = JSON.parse(result.agentOutput);
421
+ return { pass: deepEqual(rows, GOLDEN), reasoning: 'row-set comparison' };
422
+ });
423
+
424
+ test('answers the revenue query', { prompt: '...' }, async ({ agent, evaluate }) => {
425
+ const result = await agent.run();
426
+ await evaluate(result, 'sql-matches-golden'); // gate
427
+ await evaluate.observe(result, 'rows-are-sorted'); // observe (non-gating)
428
+ });
429
+ ```
430
+
431
+ - The evaluator fn receives `{ result, criteria?, traces? }` and returns
432
+ `{ pass, score?, reasoning? }`.
433
+ - It records a `MatcherResult` with `method: 'evaluator'`, **gates by default**,
434
+ and runs **in-process** — deterministic and free (no LLM call).
435
+ - `evaluate.observe(...)` feeds score/insights only, mirroring `judge.observe`.
436
+
437
+ ### Traces fixture
438
+
439
+ After `await agent.run()` resolves, OTel data is available synchronously on
440
+ `result.traces` (and on the standalone `traces` fixture for the legacy eager
441
+ path) when the agent has `useTraces: true`:
442
+
443
+ ```javascript
444
+ expect(traces.totalTokens).to.be.lessThan(10_000);
445
+ expect(traces.totalCost).to.be.lessThan(0.05);
446
+ expect(traces.spanDuration('search_logs')).to.be.lessThan(2_000);
447
+ expect(traces.toolCalls).to.have.length.greaterThan(0);
448
+ expect(traces.spans).to.have.length.greaterThan(0); // raw access for power users
449
+ ```
450
+
451
+ Availability rules:
452
+
453
+ - **Agent has `useTraces: false`** — every accessor returns `0` / `[]`. This
454
+ is the opt-out path; assertions like `traces.totalTokens === 0` are still
455
+ meaningful (they assert “I didn't expect any traces”).
456
+ - **Agent has `useTraces: true` and spans were fetched** — accessors return
457
+ the real aggregated values.
458
+ - **Agent has `useTraces: true` but spans were not retrievable** — every
459
+ read **throws** with a specific reason:
460
+ - `agent has useTraces=true but produced no runId for trace correlation`
461
+ - `fetch failed for runId=…: <underlying error message>` (transient or
462
+ persistent backend errors)
463
+ - `no spans found for runId=… after polling — verify the agent's OTel
464
+ exporter is reachable`
465
+
466
+ This turns the silent false-pass described in [#230] into an actionable
467
+ failure.
468
+
469
+ - **The body never calls `agent.run()`** (a data-only / deterministic test) —
470
+ the `traces` fixture starts *unavailable* and every read **throws**
471
+ `traces are only available after agent.run() has been called`. Traces are a
472
+ property of an agent invocation, so a body that never invokes the agent has
473
+ none to read. Call `agent.run()` first, or don't read `traces` in that test.
474
+
475
+ Polling is bounded so the test body never blocks for long: by default
476
+ 10 attempts at 1s each (~10s budget). Override per agent via the
477
+ `tracePolling.intervalMs` / `tracePolling.maxAttempts` fields, or globally
478
+ via the `TRACE_POLL_INTERVAL_MS` / `TRACE_POLL_MAX_ATTEMPTS` env vars (the
479
+ same vars the judge poller honours). A hard ceiling of 60 attempts is
480
+ enforced regardless of configuration.
481
+
482
+ [#230]: https://github.com/opensearch-project/agent-health/issues/230
483
+
484
+ ---
485
+
486
+ ## Running the tests
487
+
488
+ ### Via the UI
489
+
490
+ `/evaluations/runs/new` → pick "Code import" → select your `.eval.js` files.
491
+
492
+ ### Via the CLI
493
+
494
+ ```bash
495
+ npx @opensearch-project/agent-health benchmark -f ./evals/demo.eval.js -a observio
496
+ ```
497
+
498
+ ### Via the HTTP API
499
+
500
+ ```bash
501
+ curl -sN -X POST http://localhost:4001/api/storage/evaluation-runs \
502
+ -H 'Content-Type: application/json' \
503
+ -d '{
504
+ "name": "Demo",
505
+ "sources": [{
506
+ "type": "code-import",
507
+ "filenames": ["evals/demo.eval.js"],
508
+ "testCaseIds": []
509
+ }],
510
+ "agentKey": "observio",
511
+ "modelId": "claude-sonnet"
512
+ }'
513
+ ```
514
+
515
+ ### Migrating v1 → v2
516
+
517
+ A codemod rewrites old eager-style eval files (a destructured `result` fixture)
518
+ to the v2 control-inversion shape (`const result = await agent.run()`):
519
+
520
+ ```bash
521
+ npx @opensearch-project/agent-health migrate sdk-v2 ./evals/*.eval.js # writes in place
522
+ npx @opensearch-project/agent-health migrate sdk-v2 ./evals --dry-run # preview only
523
+ ```
524
+
525
+ The `.eval.js`, `.eval.ts`, and `.eval.mjs` loaders all run through a single
526
+ code-import execution path, so `benchmark -f <file>` executes the SDK body
527
+ directly.
528
+
529
+ ---
530
+
531
+ ## Dev tips
532
+
533
+ ### Suppress the experimental warning in your test runs
534
+
535
+ ```bash
536
+ AH_SUPPRESS_EXPERIMENTAL=1 npm test
537
+ ```
538
+
539
+ ### Get IntelliSense for custom matchers in TypeScript
540
+
541
+ Drop a tiny `chai-augmentations.d.ts` in your project:
542
+
543
+ ```typescript
544
+ // chai-augmentations.d.ts
545
+ import 'chai';
546
+ declare global {
547
+ namespace Chai {
548
+ interface Assertion {
549
+ haveCalledTool(name: string, args?: Record<string, unknown>): Assertion;
550
+ haveStepsOfType(type: string): Assertion;
551
+ haveOutputMatching(pattern: RegExp | string): Assertion;
552
+ haveCompletedWithin(ms: number): Assertion;
553
+ }
554
+ }
555
+ }
556
+ ```
557
+
558
+ We don't ship this with the SDK because chai@4's types use ambient namespaces
559
+ that conflict with other `Assertion` types in the OpenSearch ecosystem —
560
+ keeping it user-supplied lets you opt in without breaking anyone else.
561
+
562
+ ---
563
+
564
+ ## Roadmap
565
+
566
+ - [x] Optional fields on TestOptions; only `name` required
567
+ - [x] Within-file duplicate detection
568
+ - [x] No-prompt mode (skip agent invocation entirely)
569
+ - [x] Per-matcher results — chai recording plugin + judge() + traces helper
570
+ - [x] UI breakdown panel
571
+ - [x] Real traces pre-loading from OTel exporter (#230)
572
+ - [x] Lifecycle hooks (`beforeEach`/`afterEach`/`beforeAll`/`afterAll`) with `provide()` for per-test out-of-band provisioning ([#229](https://github.com/opensearch-project/agent-health/issues/229))
573
+ - [x] Control inversion — the `agent` fixture (`await agent.run()`, one invocation per test) (RFC 004 / [#256](https://github.com/opensearch-project/agent-health/issues/256))
574
+ - [x] Non-throwing run-scoped `judge` with `gate` / `observe` roles + `skip` + `orThrow()`
575
+ - [x] `defineEvaluator()` / `evaluate()` for mechanical / external verification ([#244](https://github.com/opensearch-project/agent-health/issues/244))
576
+ - [x] Single code-import execution path (`benchmark -f *.eval.js` runs the SDK body) + unified `.js` / `.ts` / `.mjs` loaders + `agent-health migrate sdk-v2` codemod
577
+ - [ ] `expect.soft()` to collect-all-failures instead of bail-on-first