@opensearch-project/agent-health 0.3.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +77 -6
- package/cli/dist/index.js +10072 -4502
- package/deployment/cloudformation/agent-health-observability.yaml +762 -0
- package/dist/assets/index-CCQRDlO0.js +243 -0
- package/dist/assets/index-CNHQVbcj.css +1 -0
- package/dist/index.html +2 -2
- package/docs/ARCHITECTURE.md +450 -0
- package/docs/BACKEND_JOB_QUEUE.md +405 -0
- package/docs/CLAUDE_CODE_TELEMETRY.md +283 -0
- package/docs/CLI.md +431 -0
- package/docs/CODING_AGENT_ANALYTICS.md +298 -0
- package/docs/CONFIGURATION.md +388 -0
- package/docs/CONNECTORS.md +536 -0
- package/docs/INSTRUMENT_WITH_OTEL.md +390 -0
- package/docs/ML-COMMONS-SETUP.md +289 -0
- package/docs/NPX_PACKAGING.md +195 -0
- package/docs/PERFORMANCE-MONITORING.md +200 -0
- package/docs/PERFORMANCE.md +390 -0
- package/docs/PI_PROFILING.md +169 -0
- package/docs/PLAN-non-agui-agent-support.md +525 -0
- package/docs/SDK.md +577 -0
- package/docs/SKILLS.md +264 -0
- package/docs/blogs/2026-02-28-opensearch-agent-health.md +200 -0
- package/docs/blogs/getting-started-blog.md +608 -0
- package/docs/diagrams/Agent-health.excalidraw +5656 -0
- package/docs/diagrams/architecture.png +0 -0
- package/docs/plans/field-redesign.md +468 -0
- package/docs/rfcs/001-coding-agent-analytics.md +374 -0
- package/docs/rfcs/002-enterprise-leaderboard.md +267 -0
- package/docs/rfcs/003-remote-aggregation.md +146 -0
- package/docs/rfcs/004-test-sdk-v2.md +599 -0
- package/docs/skills/AGENT_HEALTH.md +598 -0
- package/docs/skills/AGENT_PROFILE.md +191 -0
- package/docs/skills/add-connector/SKILL.md +68 -0
- package/docs/skills/agent-health-profile/SKILL.md +40 -0
- package/docs/skills/config-auth/SKILL.md +194 -0
- package/docs/skills/config-auth/evals/evals.json +35 -0
- package/docs/skills/create-pr/SKILL.md +73 -0
- package/docs/skills/instrument-otel/SKILL.md +84 -0
- package/docs/skills/write-test/SKILL.md +124 -0
- package/docs/ui prd.md +376 -0
- package/examples/README.md +53 -0
- package/examples/config/agent-health.config.example.ts +155 -0
- package/examples/connectors/echo-connector.ts +131 -0
- package/examples/eval-files/demo.eval.js +128 -0
- package/examples/eval-files/sdk-hooks-demo.eval.js +99 -0
- package/examples/pi-profiling/README.md +77 -0
- package/examples/pi-profiling/agent-health-profile.ts +417 -0
- package/lib/dist/lib/agentUtils.d.ts +29 -0
- package/lib/dist/lib/agentUtils.d.ts.map +1 -0
- package/lib/dist/lib/agentUtils.js +43 -0
- package/lib/dist/lib/agentUtils.js.map +1 -0
- package/lib/dist/lib/benchmarkExport.d.ts +14 -0
- package/lib/dist/lib/benchmarkExport.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkExport.js +41 -0
- package/lib/dist/lib/benchmarkExport.js.map +1 -0
- package/lib/dist/lib/benchmarkVersionUtils.d.ts +37 -0
- package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkVersionUtils.js +68 -0
- package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -0
- package/lib/dist/lib/config/defineConfig.d.ts +27 -0
- package/lib/dist/lib/config/defineConfig.d.ts.map +1 -0
- package/lib/dist/lib/config/defineConfig.js +28 -0
- package/lib/dist/lib/config/defineConfig.js.map +1 -0
- package/lib/dist/lib/config/index.d.ts +9 -0
- package/lib/dist/lib/config/index.d.ts.map +1 -0
- package/lib/dist/lib/config/index.js +8 -0
- package/lib/dist/lib/config/index.js.map +1 -0
- package/lib/dist/lib/config/loader.d.ts +39 -0
- package/lib/dist/lib/config/loader.d.ts.map +1 -0
- package/lib/dist/lib/config/loader.js +258 -0
- package/lib/dist/lib/config/loader.js.map +1 -0
- package/lib/dist/lib/config/statePaths.d.ts +61 -0
- package/lib/dist/lib/config/statePaths.d.ts.map +1 -0
- package/lib/dist/lib/config/statePaths.js +188 -0
- package/lib/dist/lib/config/statePaths.js.map +1 -0
- package/lib/dist/lib/config/types.d.ts +231 -0
- package/lib/dist/lib/config/types.d.ts.map +1 -0
- package/lib/dist/lib/config/types.js +6 -0
- package/lib/dist/lib/config/types.js.map +1 -0
- package/lib/dist/lib/config.d.ts +39 -0
- package/lib/dist/lib/config.d.ts.map +1 -0
- package/lib/dist/lib/config.js +118 -0
- package/lib/dist/lib/config.js.map +1 -0
- package/lib/dist/lib/constants.d.ts +70 -0
- package/lib/dist/lib/constants.d.ts.map +1 -0
- package/lib/dist/lib/constants.js +365 -0
- package/lib/dist/lib/constants.js.map +1 -0
- package/lib/dist/lib/contextUtilization.d.ts +23 -0
- package/lib/dist/lib/contextUtilization.d.ts.map +1 -0
- package/lib/dist/lib/contextUtilization.js +72 -0
- package/lib/dist/lib/contextUtilization.js.map +1 -0
- package/lib/dist/lib/dashboardMetrics.d.ts +87 -0
- package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -0
- package/lib/dist/lib/dashboardMetrics.js +242 -0
- package/lib/dist/lib/dashboardMetrics.js.map +1 -0
- package/lib/dist/lib/dataSourceConfig.d.ts +108 -0
- package/lib/dist/lib/dataSourceConfig.d.ts.map +1 -0
- package/lib/dist/lib/dataSourceConfig.js +166 -0
- package/lib/dist/lib/dataSourceConfig.js.map +1 -0
- package/lib/dist/lib/debug.d.ts +26 -0
- package/lib/dist/lib/debug.d.ts.map +1 -0
- package/lib/dist/lib/debug.js +132 -0
- package/lib/dist/lib/debug.js.map +1 -0
- package/lib/dist/lib/diagnostics.d.ts +28 -0
- package/lib/dist/lib/diagnostics.d.ts.map +1 -0
- package/lib/dist/lib/diagnostics.js +65 -0
- package/lib/dist/lib/diagnostics.js.map +1 -0
- package/lib/dist/lib/envCompat.d.ts +27 -0
- package/lib/dist/lib/envCompat.d.ts.map +1 -0
- package/lib/dist/lib/envCompat.js +73 -0
- package/lib/dist/lib/envCompat.js.map +1 -0
- package/lib/dist/lib/findPackageRoot.d.ts +7 -0
- package/lib/dist/lib/findPackageRoot.d.ts.map +1 -0
- package/lib/dist/lib/findPackageRoot.js +57 -0
- package/lib/dist/lib/findPackageRoot.js.map +1 -0
- package/lib/dist/lib/hooks.d.ts +36 -0
- package/lib/dist/lib/hooks.d.ts.map +1 -0
- package/lib/dist/lib/hooks.js +112 -0
- package/lib/dist/lib/hooks.js.map +1 -0
- package/lib/dist/lib/index.d.ts +47 -0
- package/lib/dist/lib/index.d.ts.map +1 -0
- package/lib/dist/lib/index.js +62 -0
- package/lib/dist/lib/index.js.map +1 -0
- package/lib/dist/lib/labels.d.ts +90 -0
- package/lib/dist/lib/labels.d.ts.map +1 -0
- package/lib/dist/lib/labels.js +158 -0
- package/lib/dist/lib/labels.js.map +1 -0
- package/lib/dist/lib/markdown.d.ts +16 -0
- package/lib/dist/lib/markdown.d.ts.map +1 -0
- package/lib/dist/lib/markdown.js +42 -0
- package/lib/dist/lib/markdown.js.map +1 -0
- package/lib/dist/lib/matchers/expect.d.ts +3 -0
- package/lib/dist/lib/matchers/expect.d.ts.map +1 -0
- package/lib/dist/lib/matchers/expect.js +225 -0
- package/lib/dist/lib/matchers/expect.js.map +1 -0
- package/lib/dist/lib/matchers/index.d.ts +8 -0
- package/lib/dist/lib/matchers/index.d.ts.map +1 -0
- package/lib/dist/lib/matchers/index.js +9 -0
- package/lib/dist/lib/matchers/index.js.map +1 -0
- package/lib/dist/lib/matchers/judgeAccessor.d.ts +113 -0
- package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -0
- package/lib/dist/lib/matchers/judgeAccessor.js +183 -0
- package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -0
- package/lib/dist/lib/matchers/session.d.ts +39 -0
- package/lib/dist/lib/matchers/session.d.ts.map +1 -0
- package/lib/dist/lib/matchers/session.js +116 -0
- package/lib/dist/lib/matchers/session.js.map +1 -0
- package/lib/dist/lib/matchers/traces.d.ts +55 -0
- package/lib/dist/lib/matchers/traces.d.ts.map +1 -0
- package/lib/dist/lib/matchers/traces.js +116 -0
- package/lib/dist/lib/matchers/traces.js.map +1 -0
- package/lib/dist/lib/matchers/types.d.ts +75 -0
- package/lib/dist/lib/matchers/types.d.ts.map +1 -0
- package/lib/dist/lib/matchers/types.js +6 -0
- package/lib/dist/lib/matchers/types.js.map +1 -0
- package/lib/dist/lib/packagePaths.d.ts +29 -0
- package/lib/dist/lib/packagePaths.d.ts.map +1 -0
- package/lib/dist/lib/packagePaths.js +63 -0
- package/lib/dist/lib/packagePaths.js.map +1 -0
- package/lib/dist/lib/performance.d.ts +51 -0
- package/lib/dist/lib/performance.d.ts.map +1 -0
- package/lib/dist/lib/performance.js +159 -0
- package/lib/dist/lib/performance.js.map +1 -0
- package/lib/dist/lib/portConfig.d.ts +29 -0
- package/lib/dist/lib/portConfig.d.ts.map +1 -0
- package/lib/dist/lib/portConfig.js +64 -0
- package/lib/dist/lib/portConfig.js.map +1 -0
- package/lib/dist/lib/preferences.d.ts +63 -0
- package/lib/dist/lib/preferences.d.ts.map +1 -0
- package/lib/dist/lib/preferences.js +117 -0
- package/lib/dist/lib/preferences.js.map +1 -0
- package/lib/dist/lib/resolveAgentModel.d.ts +22 -0
- package/lib/dist/lib/resolveAgentModel.d.ts.map +1 -0
- package/lib/dist/lib/resolveAgentModel.js +37 -0
- package/lib/dist/lib/resolveAgentModel.js.map +1 -0
- package/lib/dist/lib/runStats.d.ts +92 -0
- package/lib/dist/lib/runStats.d.ts.map +1 -0
- package/lib/dist/lib/runStats.js +160 -0
- package/lib/dist/lib/runStats.js.map +1 -0
- package/lib/dist/lib/telemetry/constants.d.ts +60 -0
- package/lib/dist/lib/telemetry/constants.d.ts.map +1 -0
- package/lib/dist/lib/telemetry/constants.js +87 -0
- package/lib/dist/lib/telemetry/constants.js.map +1 -0
- package/lib/dist/lib/telemetry/evalSpans.d.ts +61 -0
- package/lib/dist/lib/telemetry/evalSpans.d.ts.map +1 -0
- package/lib/dist/lib/telemetry/evalSpans.js +254 -0
- package/lib/dist/lib/telemetry/evalSpans.js.map +1 -0
- package/lib/dist/lib/telemetry/index.d.ts +11 -0
- package/lib/dist/lib/telemetry/index.d.ts.map +1 -0
- package/lib/dist/lib/telemetry/index.js +15 -0
- package/lib/dist/lib/telemetry/index.js.map +1 -0
- package/lib/dist/lib/telemetry/opensearchExporter.d.ts +43 -0
- package/lib/dist/lib/telemetry/opensearchExporter.d.ts.map +1 -0
- package/lib/dist/lib/telemetry/opensearchExporter.js +217 -0
- package/lib/dist/lib/telemetry/opensearchExporter.js.map +1 -0
- package/lib/dist/lib/telemetry/provider.d.ts +55 -0
- package/lib/dist/lib/telemetry/provider.d.ts.map +1 -0
- package/lib/dist/lib/telemetry/provider.js +140 -0
- package/lib/dist/lib/telemetry/provider.js.map +1 -0
- package/lib/dist/lib/testCaseLabels.d.ts +34 -0
- package/lib/dist/lib/testCaseLabels.d.ts.map +1 -0
- package/lib/dist/lib/testCaseLabels.js +88 -0
- package/lib/dist/lib/testCaseLabels.js.map +1 -0
- package/lib/dist/lib/testCaseValidation.d.ts +140 -0
- package/lib/dist/lib/testCaseValidation.d.ts.map +1 -0
- package/lib/dist/lib/testCaseValidation.js +162 -0
- package/lib/dist/lib/testCaseValidation.js.map +1 -0
- package/lib/dist/lib/testCases/agentFixture.d.ts +80 -0
- package/lib/dist/lib/testCases/agentFixture.d.ts.map +1 -0
- package/lib/dist/lib/testCases/agentFixture.js +43 -0
- package/lib/dist/lib/testCases/agentFixture.js.map +1 -0
- package/lib/dist/lib/testCases/authoringSurface.d.ts +10 -0
- package/lib/dist/lib/testCases/authoringSurface.d.ts.map +1 -0
- package/lib/dist/lib/testCases/authoringSurface.js +54 -0
- package/lib/dist/lib/testCases/authoringSurface.js.map +1 -0
- package/lib/dist/lib/testCases/codemod.d.ts +13 -0
- package/lib/dist/lib/testCases/codemod.d.ts.map +1 -0
- package/lib/dist/lib/testCases/codemod.js +169 -0
- package/lib/dist/lib/testCases/codemod.js.map +1 -0
- package/lib/dist/lib/testCases/define.d.ts +114 -0
- package/lib/dist/lib/testCases/define.d.ts.map +1 -0
- package/lib/dist/lib/testCases/define.js +253 -0
- package/lib/dist/lib/testCases/define.js.map +1 -0
- package/lib/dist/lib/testCases/evaluators.d.ts +80 -0
- package/lib/dist/lib/testCases/evaluators.d.ts.map +1 -0
- package/lib/dist/lib/testCases/evaluators.js +105 -0
- package/lib/dist/lib/testCases/evaluators.js.map +1 -0
- package/lib/dist/lib/testCases/index.d.ts +14 -0
- package/lib/dist/lib/testCases/index.d.ts.map +1 -0
- package/lib/dist/lib/testCases/index.js +12 -0
- package/lib/dist/lib/testCases/index.js.map +1 -0
- package/lib/dist/lib/testCases/judge.d.ts +165 -0
- package/lib/dist/lib/testCases/judge.d.ts.map +1 -0
- package/lib/dist/lib/testCases/judge.js +359 -0
- package/lib/dist/lib/testCases/judge.js.map +1 -0
- package/lib/dist/lib/testCases/loader.d.ts +26 -0
- package/lib/dist/lib/testCases/loader.d.ts.map +1 -0
- package/lib/dist/lib/testCases/loader.js +149 -0
- package/lib/dist/lib/testCases/loader.js.map +1 -0
- package/lib/dist/lib/testCases/types.d.ts +242 -0
- package/lib/dist/lib/testCases/types.d.ts.map +1 -0
- package/lib/dist/lib/testCases/types.js +6 -0
- package/lib/dist/lib/testCases/types.js.map +1 -0
- package/lib/dist/lib/theme.d.ts +6 -0
- package/lib/dist/lib/theme.d.ts.map +1 -0
- package/lib/dist/lib/theme.js +36 -0
- package/lib/dist/lib/theme.js.map +1 -0
- package/lib/dist/lib/uiTelemetry.d.ts +7 -0
- package/lib/dist/lib/uiTelemetry.d.ts.map +1 -0
- package/lib/dist/lib/uiTelemetry.js +25 -0
- package/lib/dist/lib/uiTelemetry.js.map +1 -0
- package/lib/dist/lib/utils.d.ts +96 -0
- package/lib/dist/lib/utils.d.ts.map +1 -0
- package/lib/dist/lib/utils.js +232 -0
- package/lib/dist/lib/utils.js.map +1 -0
- package/lib/dist/lib/workflow/consolidate.d.ts +12 -0
- package/lib/dist/lib/workflow/consolidate.d.ts.map +1 -0
- package/lib/dist/lib/workflow/consolidate.js +33 -0
- package/lib/dist/lib/workflow/consolidate.js.map +1 -0
- package/lib/dist/lib/workflow/index.d.ts +13 -0
- package/lib/dist/lib/workflow/index.d.ts.map +1 -0
- package/lib/dist/lib/workflow/index.js +12 -0
- package/lib/dist/lib/workflow/index.js.map +1 -0
- package/lib/dist/lib/workflow/ledger.d.ts +30 -0
- package/lib/dist/lib/workflow/ledger.d.ts.map +1 -0
- package/lib/dist/lib/workflow/ledger.js +41 -0
- package/lib/dist/lib/workflow/ledger.js.map +1 -0
- package/lib/dist/lib/workflow/pool.d.ts +13 -0
- package/lib/dist/lib/workflow/pool.d.ts.map +1 -0
- package/lib/dist/lib/workflow/pool.js +44 -0
- package/lib/dist/lib/workflow/pool.js.map +1 -0
- package/lib/dist/lib/workflow/source.d.ts +22 -0
- package/lib/dist/lib/workflow/source.d.ts.map +1 -0
- package/lib/dist/lib/workflow/source.js +29 -0
- package/lib/dist/lib/workflow/source.js.map +1 -0
- package/lib/dist/lib/workflow/stepB.d.ts +71 -0
- package/lib/dist/lib/workflow/stepB.d.ts.map +1 -0
- package/lib/dist/lib/workflow/stepB.js +99 -0
- package/lib/dist/lib/workflow/stepB.js.map +1 -0
- package/lib/dist/lib/workflow/types.d.ts +86 -0
- package/lib/dist/lib/workflow/types.d.ts.map +1 -0
- package/lib/dist/lib/workflow/types.js +6 -0
- package/lib/dist/lib/workflow/types.js.map +1 -0
- package/lib/dist/lib/workflow/workflow.d.ts +119 -0
- package/lib/dist/lib/workflow/workflow.d.ts.map +1 -0
- package/lib/dist/lib/workflow/workflow.js +195 -0
- package/lib/dist/lib/workflow/workflow.js.map +1 -0
- package/lib/dist/services/agent/aguiConverter.d.ts +50 -0
- package/lib/dist/services/agent/aguiConverter.d.ts.map +1 -0
- package/lib/dist/services/agent/aguiConverter.js +449 -0
- package/lib/dist/services/agent/aguiConverter.js.map +1 -0
- package/lib/dist/services/agent/index.d.ts +10 -0
- package/lib/dist/services/agent/index.d.ts.map +1 -0
- package/lib/dist/services/agent/index.js +12 -0
- package/lib/dist/services/agent/index.js.map +1 -0
- package/lib/dist/services/agent/payloadBuilder.d.ts +33 -0
- package/lib/dist/services/agent/payloadBuilder.d.ts.map +1 -0
- package/lib/dist/services/agent/payloadBuilder.js +75 -0
- package/lib/dist/services/agent/payloadBuilder.js.map +1 -0
- package/lib/dist/services/agent/sseStream.d.ts +43 -0
- package/lib/dist/services/agent/sseStream.d.ts.map +1 -0
- package/lib/dist/services/agent/sseStream.js +223 -0
- package/lib/dist/services/agent/sseStream.js.map +1 -0
- package/lib/dist/services/connectors/agui/AGUIStreamingConnector.d.ts +44 -0
- package/lib/dist/services/connectors/agui/AGUIStreamingConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/agui/AGUIStreamingConnector.js +95 -0
- package/lib/dist/services/connectors/agui/AGUIStreamingConnector.js.map +1 -0
- package/lib/dist/services/connectors/base/BaseConnector.d.ts +81 -0
- package/lib/dist/services/connectors/base/BaseConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/base/BaseConnector.js +170 -0
- package/lib/dist/services/connectors/base/BaseConnector.js.map +1 -0
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +116 -0
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +403 -0
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -0
- package/lib/dist/services/connectors/index.d.ts +13 -0
- package/lib/dist/services/connectors/index.d.ts.map +1 -0
- package/lib/dist/services/connectors/index.js +32 -0
- package/lib/dist/services/connectors/index.js.map +1 -0
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +48 -0
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/kiro/KiroConnector.js +158 -0
- package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -0
- package/lib/dist/services/connectors/langgraph/LangGraphConnector.d.ts +36 -0
- package/lib/dist/services/connectors/langgraph/LangGraphConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/langgraph/LangGraphConnector.js +175 -0
- package/lib/dist/services/connectors/langgraph/LangGraphConnector.js.map +1 -0
- package/lib/dist/services/connectors/mock/MockConnector.d.ts +37 -0
- package/lib/dist/services/connectors/mock/MockConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/mock/MockConnector.js +120 -0
- package/lib/dist/services/connectors/mock/MockConnector.js.map +1 -0
- package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.d.ts +42 -0
- package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.js +133 -0
- package/lib/dist/services/connectors/openai-compatible/OpenAICompatibleConnector.js.map +1 -0
- package/lib/dist/services/connectors/pi/PiConnector.d.ts +87 -0
- package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/pi/PiConnector.js +274 -0
- package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -0
- package/lib/dist/services/connectors/registry.d.ts +57 -0
- package/lib/dist/services/connectors/registry.d.ts.map +1 -0
- package/lib/dist/services/connectors/registry.js +106 -0
- package/lib/dist/services/connectors/registry.js.map +1 -0
- package/lib/dist/services/connectors/rest/RESTConnector.d.ts +38 -0
- package/lib/dist/services/connectors/rest/RESTConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/rest/RESTConnector.js +117 -0
- package/lib/dist/services/connectors/rest/RESTConnector.js.map +1 -0
- package/lib/dist/services/connectors/server.d.ts +13 -0
- package/lib/dist/services/connectors/server.d.ts.map +1 -0
- package/lib/dist/services/connectors/server.js +34 -0
- package/lib/dist/services/connectors/server.js.map +1 -0
- package/lib/dist/services/connectors/strands/StrandsConnector.d.ts +48 -0
- package/lib/dist/services/connectors/strands/StrandsConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/strands/StrandsConnector.js +221 -0
- package/lib/dist/services/connectors/strands/StrandsConnector.js.map +1 -0
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +88 -0
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -0
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +418 -0
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -0
- package/lib/dist/services/connectors/types.d.ts +213 -0
- package/lib/dist/services/connectors/types.d.ts.map +1 -0
- package/lib/dist/services/connectors/types.js +6 -0
- package/lib/dist/services/connectors/types.js.map +1 -0
- package/lib/dist/services/evaluation/bedrockJudge.d.ts +64 -0
- package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -0
- package/lib/dist/services/evaluation/bedrockJudge.js +167 -0
- package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -0
- package/lib/dist/services/evaluation/evaluatorError.d.ts +56 -0
- package/lib/dist/services/evaluation/evaluatorError.d.ts.map +1 -0
- package/lib/dist/services/evaluation/evaluatorError.js +56 -0
- package/lib/dist/services/evaluation/evaluatorError.js.map +1 -0
- package/lib/dist/services/evaluation/index.d.ts +106 -0
- package/lib/dist/services/evaluation/index.d.ts.map +1 -0
- package/lib/dist/services/evaluation/index.js +684 -0
- package/lib/dist/services/evaluation/index.js.map +1 -0
- package/lib/dist/services/evaluation/mockTrajectory.d.ts +3 -0
- package/lib/dist/services/evaluation/mockTrajectory.d.ts.map +1 -0
- package/lib/dist/services/evaluation/mockTrajectory.js +72 -0
- package/lib/dist/services/evaluation/mockTrajectory.js.map +1 -0
- package/lib/dist/services/opensearch/client.d.ts +26 -0
- package/lib/dist/services/opensearch/client.d.ts.map +1 -0
- package/lib/dist/services/opensearch/client.js +131 -0
- package/lib/dist/services/opensearch/client.js.map +1 -0
- package/lib/dist/services/opensearch/index.d.ts +16 -0
- package/lib/dist/services/opensearch/index.d.ts.map +1 -0
- package/lib/dist/services/opensearch/index.js +25 -0
- package/lib/dist/services/opensearch/index.js.map +1 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +123 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.js +429 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -0
- package/lib/dist/services/storage/asyncRunStorage.d.ts +127 -0
- package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -0
- package/lib/dist/services/storage/asyncRunStorage.js +448 -0
- package/lib/dist/services/storage/asyncRunStorage.js.map +1 -0
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +156 -0
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -0
- package/lib/dist/services/storage/asyncTestCaseStorage.js +285 -0
- package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -0
- package/lib/dist/services/storage/index.d.ts +17 -0
- package/lib/dist/services/storage/index.d.ts.map +1 -0
- package/lib/dist/services/storage/index.js +20 -0
- package/lib/dist/services/storage/index.js.map +1 -0
- package/lib/dist/services/storage/migration.d.ts +54 -0
- package/lib/dist/services/storage/migration.d.ts.map +1 -0
- package/lib/dist/services/storage/migration.js +296 -0
- package/lib/dist/services/storage/migration.js.map +1 -0
- package/lib/dist/services/storage/opensearchClient.d.ts +924 -0
- package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -0
- package/lib/dist/services/storage/opensearchClient.js +435 -0
- package/lib/dist/services/storage/opensearchClient.js.map +1 -0
- package/lib/dist/services/traces/browserRecovery.d.ts +26 -0
- package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -0
- package/lib/dist/services/traces/browserRecovery.js +81 -0
- package/lib/dist/services/traces/browserRecovery.js.map +1 -0
- package/lib/dist/services/traces/categoryStyles.d.ts +21 -0
- package/lib/dist/services/traces/categoryStyles.d.ts.map +1 -0
- package/lib/dist/services/traces/categoryStyles.js +56 -0
- package/lib/dist/services/traces/categoryStyles.js.map +1 -0
- package/lib/dist/services/traces/executionOrderTransform.d.ts +35 -0
- package/lib/dist/services/traces/executionOrderTransform.d.ts.map +1 -0
- package/lib/dist/services/traces/executionOrderTransform.js +313 -0
- package/lib/dist/services/traces/executionOrderTransform.js.map +1 -0
- package/lib/dist/services/traces/fetchSpansForRun.d.ts +86 -0
- package/lib/dist/services/traces/fetchSpansForRun.d.ts.map +1 -0
- package/lib/dist/services/traces/fetchSpansForRun.js +69 -0
- package/lib/dist/services/traces/fetchSpansForRun.js.map +1 -0
- package/lib/dist/services/traces/flowTransform.d.ts +24 -0
- package/lib/dist/services/traces/flowTransform.d.ts.map +1 -0
- package/lib/dist/services/traces/flowTransform.js +228 -0
- package/lib/dist/services/traces/flowTransform.js.map +1 -0
- package/lib/dist/services/traces/index.d.ts +121 -0
- package/lib/dist/services/traces/index.d.ts.map +1 -0
- package/lib/dist/services/traces/index.js +255 -0
- package/lib/dist/services/traces/index.js.map +1 -0
- package/lib/dist/services/traces/intentTransform.d.ts +20 -0
- package/lib/dist/services/traces/intentTransform.d.ts.map +1 -0
- package/lib/dist/services/traces/intentTransform.js +131 -0
- package/lib/dist/services/traces/intentTransform.js.map +1 -0
- package/lib/dist/services/traces/judgeAgentsHints.d.ts +63 -0
- package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -0
- package/lib/dist/services/traces/judgeAgentsHints.js +89 -0
- package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -0
- package/lib/dist/services/traces/messageExtraction.d.ts +15 -0
- package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -0
- package/lib/dist/services/traces/messageExtraction.js +251 -0
- package/lib/dist/services/traces/messageExtraction.js.map +1 -0
- package/lib/dist/services/traces/spanCategorization.d.ts +63 -0
- package/lib/dist/services/traces/spanCategorization.d.ts.map +1 -0
- package/lib/dist/services/traces/spanCategorization.js +276 -0
- package/lib/dist/services/traces/spanCategorization.js.map +1 -0
- package/lib/dist/services/traces/spanPreprocessing.d.ts +37 -0
- package/lib/dist/services/traces/spanPreprocessing.d.ts.map +1 -0
- package/lib/dist/services/traces/spanPreprocessing.js +102 -0
- package/lib/dist/services/traces/spanPreprocessing.js.map +1 -0
- package/lib/dist/services/traces/spansToTrajectory.d.ts +36 -0
- package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -0
- package/lib/dist/services/traces/spansToTrajectory.js +387 -0
- package/lib/dist/services/traces/spansToTrajectory.js.map +1 -0
- package/lib/dist/services/traces/toolSimilarity.d.ts +35 -0
- package/lib/dist/services/traces/toolSimilarity.d.ts.map +1 -0
- package/lib/dist/services/traces/toolSimilarity.js +203 -0
- package/lib/dist/services/traces/toolSimilarity.js.map +1 -0
- package/lib/dist/services/traces/traceComparison.d.ts +31 -0
- package/lib/dist/services/traces/traceComparison.d.ts.map +1 -0
- package/lib/dist/services/traces/traceComparison.js +318 -0
- package/lib/dist/services/traces/traceComparison.js.map +1 -0
- package/lib/dist/services/traces/traceGrouping.d.ts +19 -0
- package/lib/dist/services/traces/traceGrouping.d.ts.map +1 -0
- package/lib/dist/services/traces/traceGrouping.js +107 -0
- package/lib/dist/services/traces/traceGrouping.js.map +1 -0
- package/lib/dist/services/traces/tracePoller.d.ts +84 -0
- package/lib/dist/services/traces/tracePoller.d.ts.map +1 -0
- package/lib/dist/services/traces/tracePoller.js +309 -0
- package/lib/dist/services/traces/tracePoller.js.map +1 -0
- package/lib/dist/services/traces/traceStats.d.ts +45 -0
- package/lib/dist/services/traces/traceStats.d.ts.map +1 -0
- package/lib/dist/services/traces/traceStats.js +114 -0
- package/lib/dist/services/traces/traceStats.js.map +1 -0
- package/lib/dist/services/traces/traceSummary.d.ts +47 -0
- package/lib/dist/services/traces/traceSummary.d.ts.map +1 -0
- package/lib/dist/services/traces/traceSummary.js +68 -0
- package/lib/dist/services/traces/traceSummary.js.map +1 -0
- package/lib/dist/services/traces/utils.d.ts +33 -0
- package/lib/dist/services/traces/utils.d.ts.map +1 -0
- package/lib/dist/services/traces/utils.js +114 -0
- package/lib/dist/services/traces/utils.js.map +1 -0
- package/lib/dist/types/agui.d.ts +13 -0
- package/lib/dist/types/agui.d.ts.map +1 -0
- package/lib/dist/types/agui.js +16 -0
- package/lib/dist/types/agui.js.map +1 -0
- package/lib/dist/types/index.d.ts +1175 -0
- package/lib/dist/types/index.d.ts.map +1 -0
- package/lib/dist/types/index.js +12 -0
- package/lib/dist/types/index.js.map +1 -0
- package/lib/dist/types/skills.d.ts +146 -0
- package/lib/dist/types/skills.d.ts.map +1 -0
- package/lib/dist/types/skills.js +6 -0
- package/lib/dist/types/skills.js.map +1 -0
- package/observio-sample-agent/pi-package/README.md +112 -0
- package/observio-sample-agent/pi-package/extensions/agent-health.ts +373 -0
- package/observio-sample-agent/pi-package/package.json +17 -0
- package/observio-sample-agent/pi-package/prompts/agent-health.md +37 -0
- package/observio-sample-agent/pi-package/skills/create-pr/SKILL.md +88 -0
- package/observio-sample-agent/pi-package/skills/fix-bug/SKILL.md +71 -0
- package/observio-sample-agent/pi-package/skills/implement-feature/SKILL.md +156 -0
- package/observio-sample-agent/pi-package/skills/instrument-otel/SKILL.md +208 -0
- package/observio-sample-agent/pi-package/skills/setup-collector/SKILL.md +146 -0
- package/observio-sample-agent/pi-package/skills/write-test/SKILL.md +115 -0
- package/package.json +64 -13
- package/server/dist/app.js +32651 -17637
- package/server/dist/index.js +29875 -14638
- package/tsconfig.lib.json +71 -0
- package/dist/assets/index-EvPLSTAS.js +0 -267
- package/dist/assets/index-RXasQKUs.css +0 -1
- package/lib/dist/config/index.js +0 -404
- package/lib/dist/index.js +0 -1665
|
@@ -0,0 +1,684 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Evaluation Service
|
|
7
|
+
* Main orchestrator for running agent evaluations
|
|
8
|
+
*/
|
|
9
|
+
import { v4 as uuidv4 } from 'uuid';
|
|
10
|
+
import { executeBeforeRequestHook, executeAfterResponseHook } from '../../lib/hooks.js';
|
|
11
|
+
import { AGUIToTrajectoryConverter, consumeSSEStream, buildAgentPayload } from '../../services/agent/index.js';
|
|
12
|
+
import { generateMockTrajectory } from './mockTrajectory.js';
|
|
13
|
+
import { callBedrockJudge } from './bedrockJudge.js';
|
|
14
|
+
import { buildJudgeMatcherEntry, formatExpectedOutcomesAsClaim } from '../../lib/matchers/judgeAccessor.js';
|
|
15
|
+
import { buildJudgeAgentsHints } from '../../services/traces/judgeAgentsHints.js';
|
|
16
|
+
// Re-export for use by experimentRunner when calling judge after trace polling
|
|
17
|
+
export { callBedrockJudge };
|
|
18
|
+
import { openSearchClient } from '../../services/opensearch/index.js';
|
|
19
|
+
import { debug } from '../../lib/debug.js';
|
|
20
|
+
import { ENV_CONFIG } from '../../lib/config.js';
|
|
21
|
+
import { DEFAULT_CONFIG } from '../../lib/constants.js';
|
|
22
|
+
// For browser/Vite builds, use DEFAULT_CONFIG directly.
|
|
23
|
+
// For server/CLI (Node.js), we can use loadConfigSync from lib/config/index.
|
|
24
|
+
// This is a runtime check to avoid importing Node.js modules in browser builds.
|
|
25
|
+
const getModels = () => {
|
|
26
|
+
// Check if we're in a Node.js environment with file system access
|
|
27
|
+
if (typeof window === 'undefined' && typeof process !== 'undefined' && process.versions?.node) {
|
|
28
|
+
try {
|
|
29
|
+
// Dynamic import to avoid bundling issues
|
|
30
|
+
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
|
31
|
+
const { loadConfigSync } = require('../../lib/config/index.js');
|
|
32
|
+
return loadConfigSync().models;
|
|
33
|
+
}
|
|
34
|
+
catch {
|
|
35
|
+
// Fall back to defaults
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
return DEFAULT_CONFIG.models;
|
|
39
|
+
};
|
|
40
|
+
// Toggle between mock and real agent
|
|
41
|
+
const USE_MOCK_AGENT = false;
|
|
42
|
+
/**
|
|
43
|
+
* Build ConnectorAuth from AgentConfig.
|
|
44
|
+
* Prefers explicit `auth` field if present, falls back to header inference.
|
|
45
|
+
*/
|
|
46
|
+
function buildConnectorAuth(agent) {
|
|
47
|
+
// Prefer explicit auth config (new pattern)
|
|
48
|
+
if (agent.auth && agent.auth.type !== 'none') {
|
|
49
|
+
return {
|
|
50
|
+
...agent.auth,
|
|
51
|
+
// Merge any extra headers on top
|
|
52
|
+
headers: { ...agent.headers, ...agent.auth.headers },
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
// Legacy: infer from headers
|
|
56
|
+
const headers = agent.headers || {};
|
|
57
|
+
if (headers['Authorization']?.startsWith('Bearer ')) {
|
|
58
|
+
return {
|
|
59
|
+
type: 'bearer',
|
|
60
|
+
token: headers['Authorization'].replace('Bearer ', ''),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
if (headers['Authorization']?.startsWith('Basic ')) {
|
|
64
|
+
return {
|
|
65
|
+
type: 'basic',
|
|
66
|
+
token: headers['Authorization'].replace('Basic ', ''),
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
if (headers['x-api-key']) {
|
|
70
|
+
return {
|
|
71
|
+
type: 'api-key',
|
|
72
|
+
token: headers['x-api-key'],
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
// Pass through all headers
|
|
76
|
+
return {
|
|
77
|
+
type: 'none',
|
|
78
|
+
headers,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
// ─── SDK matcher-session report-level metrics shim ─────────────────────────────
|
|
82
|
+
//
|
|
83
|
+
// SDK runs (.bench.js / .eval.js with a code body) collect per-matcher
|
|
84
|
+
// verdicts in `report.matcherResults` and the runner then writes a
|
|
85
|
+
// report-level `report.metrics` with the legacy 4-key shape
|
|
86
|
+
// `{accuracy, faithfulness, latency_score, trajectory_alignment_score}` so
|
|
87
|
+
// older list/aggregate UIs keep working.
|
|
88
|
+
//
|
|
89
|
+
// Pre-fix this shim was a hardcoded `{0,0,0,0}` (failed) / `{100,100,100,100}`
|
|
90
|
+
// (passed). Two consequences:
|
|
91
|
+
// 1. A 5-of-6-passing run was indistinguishable from a 0-of-6 run — every
|
|
92
|
+
// failing run looked identical in the metrics tile. (Only meaningful
|
|
93
|
+
// for `judge()` matchers, which are non-throwing per RFC 004 §4.4 —
|
|
94
|
+
// `expect()` is fail-fast and chai-throw goes through the evalError
|
|
95
|
+
// branch below.)
|
|
96
|
+
// 2. A custom evaluator that emits per-claim dimensional scores via
|
|
97
|
+
// `MatcherResult.judgeMetrics` (e.g. a multi-dimensional RCA rubric with
|
|
98
|
+
// `routing_accuracy` / `tool_correctness` / `diagnostic_completeness`)
|
|
99
|
+
// had its dimensions silently dropped at the report-level boundary.
|
|
100
|
+
//
|
|
101
|
+
// Post-fix:
|
|
102
|
+
// - Aggregate `accuracy` is the percentage of GATE matchers that passed
|
|
103
|
+
// (rounded to integer 0..100). `observe`-role matchers and `errored`
|
|
104
|
+
// matchers are excluded from the denominator (RFC 004 §4.8).
|
|
105
|
+
// - The other three legacy keys mirror `accuracy` (BC; consumers reading
|
|
106
|
+
// them historically got 0/100, they now get a meaningful integer).
|
|
107
|
+
// - If any matcher emitted `judgeMetrics` (per-claim dimensions, see
|
|
108
|
+
// `lib/matchers/types.ts`), each dimension's MEAN across emitting gate
|
|
109
|
+
// matchers overrides the BC stub for THAT key. So a 9-dimension custom
|
|
110
|
+
// rubric ends up on `report.metrics` with `routing_accuracy: 87`,
|
|
111
|
+
// `tool_correctness: 92`, etc., not flattened to `accuracy: <yes/no>`.
|
|
112
|
+
//
|
|
113
|
+
// `evalError` (the bench body itself threw — includes a chai `expect()`
|
|
114
|
+
// fail-fast assertion) bypasses the aggregate and returns `{0,0,0,0}` — the
|
|
115
|
+
// run never reached its conclusion so a partial aggregate is misleading.
|
|
116
|
+
export function computeSdkMatcherSessionMetrics(matcherResults, opts) {
|
|
117
|
+
if (opts?.hasEvalError) {
|
|
118
|
+
return {
|
|
119
|
+
accuracy: 0,
|
|
120
|
+
faithfulness: 0,
|
|
121
|
+
latency_score: 0,
|
|
122
|
+
trajectory_alignment_score: 0,
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
const gates = matcherResults.filter(m => m.role !== 'observe' && !m.errored);
|
|
126
|
+
const total = gates.length;
|
|
127
|
+
const passing = gates.filter(m => m.pass).length;
|
|
128
|
+
// Vacuous pass when there are no gates (e.g. body just calls the agent
|
|
129
|
+
// without any judge/expect claims, or every matcher is `observe`-role).
|
|
130
|
+
// The pre-fix code returned 100 here via the `!failed` branch; keep that.
|
|
131
|
+
// When some gates DID fail, return the pass-rate percentage — the actual
|
|
132
|
+
// post-fix improvement, since pre-fix this collapsed to a flat 0.
|
|
133
|
+
const passRatePct = total === 0 ? 100 : Math.round((passing / total) * 100);
|
|
134
|
+
// BC stub — same number for all four legacy keys.
|
|
135
|
+
const out = {
|
|
136
|
+
accuracy: passRatePct,
|
|
137
|
+
faithfulness: passRatePct,
|
|
138
|
+
latency_score: passRatePct,
|
|
139
|
+
trajectory_alignment_score: passRatePct,
|
|
140
|
+
};
|
|
141
|
+
// Dimensional pass-through: aggregate any per-matcher `judgeMetrics`
|
|
142
|
+
// dimensions across emitting gates. Each dimension's mean (rounded) wins
|
|
143
|
+
// over the BC stub for the same key.
|
|
144
|
+
const sums = Object.create(null);
|
|
145
|
+
const counts = Object.create(null);
|
|
146
|
+
for (const m of gates) {
|
|
147
|
+
const dims = m.judgeMetrics;
|
|
148
|
+
if (!dims)
|
|
149
|
+
continue;
|
|
150
|
+
for (const [k, v] of Object.entries(dims)) {
|
|
151
|
+
if (typeof v === 'number' && Number.isFinite(v)) {
|
|
152
|
+
sums[k] = (sums[k] ?? 0) + v;
|
|
153
|
+
counts[k] = (counts[k] ?? 0) + 1;
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
for (const k of Object.keys(sums)) {
|
|
158
|
+
out[k] = Math.round(sums[k] / counts[k]);
|
|
159
|
+
}
|
|
160
|
+
return out;
|
|
161
|
+
}
|
|
162
|
+
/**
|
|
163
|
+
* Drive a single agent invocation through its connector and return the raw
|
|
164
|
+
* trajectory/runId/rawEvents — no judge, no report synthesis.
|
|
165
|
+
*
|
|
166
|
+
* Owns the connector resolution, request building, auth, the
|
|
167
|
+
* `beforeRequest`/`afterResponse` hooks, and execution timing. Extracted from
|
|
168
|
+
* {@link runEvaluationWithConnector} so the RFC-004 engine can offer an
|
|
169
|
+
* `agent.run()` fixture that captures exactly the same trajectory/trace
|
|
170
|
+
* correlation the legacy path produces (see RFC 004 §4.1, #256).
|
|
171
|
+
*/
|
|
172
|
+
export async function invokeAgent(agent, modelId, testCase, options) {
|
|
173
|
+
const { registry: connectorRegistry, onStep, onRawEvent, env: runEnv } = options;
|
|
174
|
+
// Get connector for this agent
|
|
175
|
+
const agentWithConnector = agent;
|
|
176
|
+
const connector = connectorRegistry.getForAgent(agentWithConnector);
|
|
177
|
+
// Merge any per-invocation env (from the SDK's AgentRunOptions.env) into the
|
|
178
|
+
// connector config so subprocess connectors forward it to the spawned child.
|
|
179
|
+
// Static config env stays the base; per-call env wins on key collisions.
|
|
180
|
+
const baseConnectorConfig = agentWithConnector.connectorConfig;
|
|
181
|
+
const mergedConnectorConfig = runEnv && Object.keys(runEnv).length > 0
|
|
182
|
+
? {
|
|
183
|
+
...(baseConnectorConfig || {}),
|
|
184
|
+
env: { ...(baseConnectorConfig?.env || {}), ...runEnv },
|
|
185
|
+
}
|
|
186
|
+
: baseConnectorConfig;
|
|
187
|
+
// Build connector request
|
|
188
|
+
let request = {
|
|
189
|
+
testCase,
|
|
190
|
+
modelId,
|
|
191
|
+
connectorConfig: mergedConnectorConfig,
|
|
192
|
+
};
|
|
193
|
+
// Build auth from agent config
|
|
194
|
+
const auth = buildConnectorAuth(agent);
|
|
195
|
+
// Execute beforeRequest hook if defined
|
|
196
|
+
let effectiveEndpoint = agent.endpoint;
|
|
197
|
+
if (agent.hooks?.beforeRequest) {
|
|
198
|
+
const previewPayload = connector.buildPayload(request);
|
|
199
|
+
const hookContext = {
|
|
200
|
+
endpoint: agent.endpoint,
|
|
201
|
+
payload: previewPayload,
|
|
202
|
+
headers: auth.headers || agent.headers || {},
|
|
203
|
+
};
|
|
204
|
+
const hookResult = await executeBeforeRequestHook(agent.hooks, hookContext, agent.key);
|
|
205
|
+
effectiveEndpoint = hookResult.endpoint;
|
|
206
|
+
// Pass the hook-modified payload through to the connector so it skips
|
|
207
|
+
// its internal buildPayload() call. This preserves ALL modifications the
|
|
208
|
+
// hook made to the payload (threadId, runId, custom fields, etc.)
|
|
209
|
+
request = {
|
|
210
|
+
...request,
|
|
211
|
+
payload: hookResult.payload,
|
|
212
|
+
};
|
|
213
|
+
// Merge any hook-modified headers into auth
|
|
214
|
+
if (hookResult.headers) {
|
|
215
|
+
auth.headers = { ...auth.headers, ...hookResult.headers };
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
// Execute via connector (with timing)
|
|
219
|
+
const agentStartTime = Date.now();
|
|
220
|
+
let result = await connector.execute(effectiveEndpoint, request, auth, onStep, onRawEvent);
|
|
221
|
+
const agentDurationMs = Date.now() - agentStartTime;
|
|
222
|
+
// Execute afterResponse hook if defined
|
|
223
|
+
if (agent.hooks?.afterResponse) {
|
|
224
|
+
debug('Eval', `Executing afterResponse hook for agent "${agent.key}"`);
|
|
225
|
+
try {
|
|
226
|
+
const hookContext = {
|
|
227
|
+
response: (result.rawEvents?.length ? result.rawEvents[result.rawEvents.length - 1] : null) || result.metadata || {},
|
|
228
|
+
trajectory: result.trajectory,
|
|
229
|
+
runId: result.runId || undefined,
|
|
230
|
+
rawEvents: result.rawEvents || [],
|
|
231
|
+
metadata: result.metadata,
|
|
232
|
+
};
|
|
233
|
+
const hookResult = await executeAfterResponseHook(agent.hooks, hookContext, agent.key);
|
|
234
|
+
// Apply hook modifications
|
|
235
|
+
result = {
|
|
236
|
+
...result,
|
|
237
|
+
trajectory: hookResult.trajectory,
|
|
238
|
+
runId: hookResult.runId || result.runId,
|
|
239
|
+
};
|
|
240
|
+
debug('Eval', 'afterResponse hook applied:', {
|
|
241
|
+
trajectorySteps: hookResult.trajectory.length,
|
|
242
|
+
runId: hookResult.runId
|
|
243
|
+
});
|
|
244
|
+
}
|
|
245
|
+
catch (hookError) {
|
|
246
|
+
const errorMsg = hookError instanceof Error ? hookError.message : String(hookError);
|
|
247
|
+
console.error(`[Eval] afterResponse hook failed for agent "${agent.key}":`, errorMsg);
|
|
248
|
+
debug('Eval', `afterResponse hook error details:`, {
|
|
249
|
+
agent: agent.key,
|
|
250
|
+
error: errorMsg,
|
|
251
|
+
rawEventsCount: result.rawEvents?.length ?? 0,
|
|
252
|
+
hasMetadata: !!result.metadata,
|
|
253
|
+
});
|
|
254
|
+
// Re-throw so the caller knows the hook failed — don't silently swallow
|
|
255
|
+
throw hookError instanceof Error ? hookError : new Error(errorMsg);
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
return {
|
|
259
|
+
trajectory: result.trajectory,
|
|
260
|
+
runId: result.runId,
|
|
261
|
+
rawEvents: result.rawEvents || [],
|
|
262
|
+
agentDurationMs,
|
|
263
|
+
metadata: result.metadata,
|
|
264
|
+
connector,
|
|
265
|
+
};
|
|
266
|
+
}
|
|
267
|
+
export async function runEvaluationWithConnector(agent, modelId, testCase, onStep, options) {
|
|
268
|
+
const { registry: connectorRegistry, onRawEvent, evaluatorId, skipJudge } = options;
|
|
269
|
+
// Pull `judgeModelId` once so the closure inside the standard-mode call
|
|
270
|
+
// below uses the run-level value rather than the agent's `modelId`.
|
|
271
|
+
const optionsJudgeModelId = options.judgeModelId;
|
|
272
|
+
const reportId = uuidv4();
|
|
273
|
+
let fullTrajectory = [];
|
|
274
|
+
let rawEvents = [];
|
|
275
|
+
let agentRunId = null;
|
|
276
|
+
let agentSessionId;
|
|
277
|
+
debug('Eval', 'Config:', { agent: agent.name, model: modelId, testCase: testCase.id });
|
|
278
|
+
const evalStartTime = Date.now();
|
|
279
|
+
try {
|
|
280
|
+
// Drive the agent through its connector (connector resolution, request
|
|
281
|
+
// building, auth, before/afterResponse hooks, timing) via the shared
|
|
282
|
+
// primitive. Judge + report synthesis stay here in the legacy path.
|
|
283
|
+
const invocation = await invokeAgent(agent, modelId, testCase, {
|
|
284
|
+
registry: connectorRegistry,
|
|
285
|
+
onStep,
|
|
286
|
+
onRawEvent,
|
|
287
|
+
});
|
|
288
|
+
const connector = invocation.connector;
|
|
289
|
+
const agentDurationMs = invocation.agentDurationMs;
|
|
290
|
+
fullTrajectory = invocation.trajectory;
|
|
291
|
+
agentRunId = invocation.runId;
|
|
292
|
+
agentSessionId = invocation.metadata?.sessionId ?? undefined;
|
|
293
|
+
rawEvents = invocation.rawEvents;
|
|
294
|
+
debug('Eval', 'Trajectory captured:', fullTrajectory.length, 'steps');
|
|
295
|
+
debug('Eval', 'Raw events captured:', rawEvents.length);
|
|
296
|
+
// TRACE MODE: Skip logs fetch and judge, return pending report
|
|
297
|
+
if (agent.useTraces) {
|
|
298
|
+
return {
|
|
299
|
+
id: reportId,
|
|
300
|
+
timestamp: new Date().toISOString(),
|
|
301
|
+
agentName: agent.name,
|
|
302
|
+
agentKey: agent.key,
|
|
303
|
+
modelName: modelId,
|
|
304
|
+
modelId: modelId,
|
|
305
|
+
testCaseId: testCase.id,
|
|
306
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
307
|
+
status: 'completed',
|
|
308
|
+
metricsStatus: 'pending',
|
|
309
|
+
trajectory: fullTrajectory,
|
|
310
|
+
metrics: {
|
|
311
|
+
accuracy: 0,
|
|
312
|
+
faithfulness: 0,
|
|
313
|
+
latency_score: 0,
|
|
314
|
+
trajectory_alignment_score: 0,
|
|
315
|
+
},
|
|
316
|
+
llmJudgeReasoning: 'Waiting for traces to become available...',
|
|
317
|
+
improvementStrategies: [],
|
|
318
|
+
runId: agentRunId || undefined,
|
|
319
|
+
sessionId: agentSessionId || undefined,
|
|
320
|
+
rawEvents,
|
|
321
|
+
connectorProtocol: connector.type,
|
|
322
|
+
performanceMetrics: {
|
|
323
|
+
durationMs: Date.now() - evalStartTime,
|
|
324
|
+
agentDurationMs,
|
|
325
|
+
},
|
|
326
|
+
};
|
|
327
|
+
}
|
|
328
|
+
// SKIP JUDGE MODE: Return report without judge evaluation (caller handles it)
|
|
329
|
+
if (skipJudge) {
|
|
330
|
+
return {
|
|
331
|
+
id: reportId,
|
|
332
|
+
timestamp: new Date().toISOString(),
|
|
333
|
+
agentName: agent.name,
|
|
334
|
+
agentKey: agent.key,
|
|
335
|
+
modelName: modelId,
|
|
336
|
+
modelId,
|
|
337
|
+
testCaseId: testCase.id,
|
|
338
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
339
|
+
status: 'completed',
|
|
340
|
+
trajectory: fullTrajectory,
|
|
341
|
+
metrics: { accuracy: 0, faithfulness: 0, latency_score: 0, trajectory_alignment_score: 0 },
|
|
342
|
+
llmJudgeReasoning: '',
|
|
343
|
+
improvementStrategies: [],
|
|
344
|
+
runId: agentRunId || undefined,
|
|
345
|
+
sessionId: agentSessionId || undefined,
|
|
346
|
+
rawEvents,
|
|
347
|
+
connectorProtocol: connector.type,
|
|
348
|
+
performanceMetrics: {
|
|
349
|
+
durationMs: Date.now() - evalStartTime,
|
|
350
|
+
agentDurationMs,
|
|
351
|
+
},
|
|
352
|
+
};
|
|
353
|
+
}
|
|
354
|
+
// STANDARD MODE: Call judge
|
|
355
|
+
const models = getModels();
|
|
356
|
+
const modelConfig = models[modelId];
|
|
357
|
+
// Judge model resolution — SEPARATE from the agent's `modelId`. Priority:
|
|
358
|
+
// 1. options.judgeModelId (run-level customer input)
|
|
359
|
+
// 2. server default `BEDROCK_MODEL_ID` (env, falls back to a recent Claude)
|
|
360
|
+
// 3. agent's modelId (last-resort BC fallback for old callers that didn't
|
|
361
|
+
// split agent vs judge model; tolerated for one release
|
|
362
|
+
// so behavior of pre-existing benchmark runs is
|
|
363
|
+
// preserved when neither cx input nor server default
|
|
364
|
+
// is available)
|
|
365
|
+
// Anything else (evaluator's own `inferenceConfig.modelId`, agentic-provider
|
|
366
|
+
// model picking) is resolved server-side in /api/judge from the evaluator.
|
|
367
|
+
const judgeModelId = optionsJudgeModelId ||
|
|
368
|
+
process.env.BEDROCK_MODEL_ID ||
|
|
369
|
+
modelConfig?.model_id ||
|
|
370
|
+
modelId;
|
|
371
|
+
const judgment = await callBedrockJudge(fullTrajectory, {
|
|
372
|
+
expectedOutcomes: testCase.expectedOutcomes,
|
|
373
|
+
expectedTrajectory: testCase.expectedTrajectory,
|
|
374
|
+
}, undefined, // No logs in direct connector mode
|
|
375
|
+
(chunk) => debug('Eval', 'Judge progress:', chunk.slice(0, 100)), judgeModelId, evaluatorId,
|
|
376
|
+
// Forward agent runId so the `agent` (trace) judge provider can
|
|
377
|
+
// scope its query_spans/query_logs tools. See callBedrockJudge.
|
|
378
|
+
agentRunId || undefined,
|
|
379
|
+
// Strategy C correlation hints (#264) so the trace judge tool can
|
|
380
|
+
// find spans the agent emits under its OWN correlation (claude-code
|
|
381
|
+
// session ids etc.), not just spans matching agent-health's runId
|
|
382
|
+
// via gen_ai.request.id.
|
|
383
|
+
buildJudgeAgentsHints({
|
|
384
|
+
agentKey: agent.key,
|
|
385
|
+
connectorProtocol: agent.connectorType,
|
|
386
|
+
timestamp: new Date().toISOString(),
|
|
387
|
+
performanceMetrics: { durationMs: Date.now() - evalStartTime, agentDurationMs },
|
|
388
|
+
}, agent.traceServiceName));
|
|
389
|
+
debug('Eval', 'Metrics:', judgment.metrics);
|
|
390
|
+
const llmJudgeResponse = {
|
|
391
|
+
modelId: judgeModelId,
|
|
392
|
+
timestamp: new Date().toISOString(),
|
|
393
|
+
promptTokens: 0,
|
|
394
|
+
completionTokens: 0,
|
|
395
|
+
latencyMs: judgment.judgeDurationMs ?? 0,
|
|
396
|
+
// Prefer the actual unparsed model output when present (post
|
|
397
|
+
// evaluator-prompt-plumbing). Pre-fix code stuffed the parsed
|
|
398
|
+
// reasoning here — fall back to that for back-compat with judges
|
|
399
|
+
// that don't yet forward the raw text.
|
|
400
|
+
rawResponse: judgment.rawResponse ?? judgment.llmJudgeReasoning,
|
|
401
|
+
parsedMetrics: judgment.metrics,
|
|
402
|
+
improvementStrategies: judgment.improvementStrategies,
|
|
403
|
+
...(judgment.extraFields ? { extraFields: judgment.extraFields } : {}),
|
|
404
|
+
...(judgment.judgeDebug ? { judgeDebug: judgment.judgeDebug } : {}),
|
|
405
|
+
};
|
|
406
|
+
return {
|
|
407
|
+
id: reportId,
|
|
408
|
+
timestamp: new Date().toISOString(),
|
|
409
|
+
agentName: agent.name,
|
|
410
|
+
agentKey: agent.key,
|
|
411
|
+
modelName: modelId,
|
|
412
|
+
modelId,
|
|
413
|
+
testCaseId: testCase.id,
|
|
414
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
415
|
+
status: 'completed',
|
|
416
|
+
passFailStatus: judgment.passFailStatus,
|
|
417
|
+
trajectory: fullTrajectory,
|
|
418
|
+
metrics: judgment.metrics,
|
|
419
|
+
llmJudgeReasoning: judgment.llmJudgeReasoning,
|
|
420
|
+
// Unified judge surface (Option-B BC: legacy field above kept).
|
|
421
|
+
matcherResults: [
|
|
422
|
+
buildJudgeMatcherEntry(judgment, {
|
|
423
|
+
claim: formatExpectedOutcomesAsClaim(testCase.expectedOutcomes),
|
|
424
|
+
model: judgeModelId,
|
|
425
|
+
}),
|
|
426
|
+
],
|
|
427
|
+
improvementStrategies: judgment.improvementStrategies,
|
|
428
|
+
llmJudgeResponse,
|
|
429
|
+
runId: agentRunId || undefined,
|
|
430
|
+
sessionId: agentSessionId || undefined,
|
|
431
|
+
rawEvents,
|
|
432
|
+
connectorProtocol: connector.type,
|
|
433
|
+
performanceMetrics: {
|
|
434
|
+
durationMs: Date.now() - evalStartTime,
|
|
435
|
+
agentDurationMs,
|
|
436
|
+
judgeDurationMs: judgment.judgeDurationMs,
|
|
437
|
+
judgeAttempts: judgment.judgeAttempts,
|
|
438
|
+
},
|
|
439
|
+
};
|
|
440
|
+
}
|
|
441
|
+
catch (error) {
|
|
442
|
+
console.error('[Eval] Error:', error instanceof Error ? error.message : error);
|
|
443
|
+
// Enhanced debug logging for connection failures
|
|
444
|
+
if (error instanceof Error) {
|
|
445
|
+
debug('Eval', 'Error details:', {
|
|
446
|
+
name: error.name,
|
|
447
|
+
message: error.message,
|
|
448
|
+
stack: error.stack,
|
|
449
|
+
cause: error.cause,
|
|
450
|
+
agent: agent.name,
|
|
451
|
+
endpoint: agent.endpoint,
|
|
452
|
+
modelId,
|
|
453
|
+
testCaseId: testCase.id,
|
|
454
|
+
});
|
|
455
|
+
}
|
|
456
|
+
else {
|
|
457
|
+
debug('Eval', 'Unknown error:', error);
|
|
458
|
+
}
|
|
459
|
+
// Get connector type for error case (may not be available if error was in getting connector)
|
|
460
|
+
let connectorType;
|
|
461
|
+
try {
|
|
462
|
+
const agentWithConnector = agent;
|
|
463
|
+
const connector = connectorRegistry.getForAgent(agentWithConnector);
|
|
464
|
+
connectorType = connector.type;
|
|
465
|
+
debug('Eval', 'Connector type:', connectorType);
|
|
466
|
+
}
|
|
467
|
+
catch (connectorError) {
|
|
468
|
+
debug('Eval', 'Failed to get connector type:', connectorError);
|
|
469
|
+
// Connector lookup failed, leave undefined
|
|
470
|
+
}
|
|
471
|
+
return {
|
|
472
|
+
id: reportId,
|
|
473
|
+
timestamp: new Date().toISOString(),
|
|
474
|
+
agentName: agent.name,
|
|
475
|
+
agentKey: agent.key,
|
|
476
|
+
modelName: modelId,
|
|
477
|
+
modelId,
|
|
478
|
+
testCaseId: testCase.id,
|
|
479
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
480
|
+
status: 'failed',
|
|
481
|
+
trajectory: fullTrajectory,
|
|
482
|
+
metrics: {
|
|
483
|
+
accuracy: 0,
|
|
484
|
+
faithfulness: 0,
|
|
485
|
+
latency_score: 0,
|
|
486
|
+
trajectory_alignment_score: 0,
|
|
487
|
+
},
|
|
488
|
+
llmJudgeReasoning: `Evaluation failed: ${error instanceof Error ? error.message : 'Unknown error'}`,
|
|
489
|
+
improvementStrategies: [],
|
|
490
|
+
rawEvents,
|
|
491
|
+
connectorProtocol: connectorType,
|
|
492
|
+
};
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
/**
|
|
496
|
+
* Run real agent evaluation by streaming AG UI events
|
|
497
|
+
* @deprecated Use runEvaluationWithConnector() via the server's /api/evaluate endpoint instead.
|
|
498
|
+
* This browser-side path is kept for backwards compatibility but has no active callers.
|
|
499
|
+
*/
|
|
500
|
+
async function runRealAgentEvaluation(agent, modelId, testCase, onStep, onRawEvent) {
|
|
501
|
+
const trajectory = [];
|
|
502
|
+
const rawEvents = [];
|
|
503
|
+
const converter = new AGUIToTrajectoryConverter();
|
|
504
|
+
const agentPayload = buildAgentPayload(testCase, modelId);
|
|
505
|
+
debug('Eval', 'Agent payload:', JSON.stringify(agentPayload).substring(0, 500));
|
|
506
|
+
// Use proxy to avoid CORS issues when calling agent endpoint
|
|
507
|
+
const proxyPayload = {
|
|
508
|
+
endpoint: agent.endpoint,
|
|
509
|
+
payload: agentPayload,
|
|
510
|
+
headers: agent.headers,
|
|
511
|
+
agentKey: agent.key,
|
|
512
|
+
};
|
|
513
|
+
debug('Eval', 'Using proxy:', ENV_CONFIG.agentProxyUrl);
|
|
514
|
+
await consumeSSEStream(ENV_CONFIG.agentProxyUrl, proxyPayload, (event) => {
|
|
515
|
+
// Capture raw event for debugging
|
|
516
|
+
rawEvents.push(event);
|
|
517
|
+
onRawEvent?.(event);
|
|
518
|
+
// Convert to trajectory steps
|
|
519
|
+
const steps = converter.processEvent(event);
|
|
520
|
+
steps.forEach(step => {
|
|
521
|
+
trajectory.push(step);
|
|
522
|
+
onStep(step);
|
|
523
|
+
});
|
|
524
|
+
});
|
|
525
|
+
const runId = converter.getRunId();
|
|
526
|
+
return { trajectory, runId, rawEvents };
|
|
527
|
+
}
|
|
528
|
+
/**
|
|
529
|
+
* Run evaluation with selected agent, model, and test case
|
|
530
|
+
* Streams trajectory steps to UI in real-time via onStep callback
|
|
531
|
+
*
|
|
532
|
+
* @deprecated Use runServerEvaluation() from services/client/evaluationApi instead.
|
|
533
|
+
* The server-side path (/api/evaluate) consolidates all evaluation logic through the
|
|
534
|
+
* connector system, ensuring consistent behavior for hooks, storage, and all agent types.
|
|
535
|
+
*/
|
|
536
|
+
export async function runEvaluation(agent, modelId, testCase, onStep, onRawEvent) {
|
|
537
|
+
const reportId = uuidv4();
|
|
538
|
+
let fullTrajectory = [];
|
|
539
|
+
let rawEvents = [];
|
|
540
|
+
let agentRunId = null;
|
|
541
|
+
let agentSessionId;
|
|
542
|
+
debug('Eval', 'Config:', { agent: agent.name, model: modelId, testCase: testCase.id });
|
|
543
|
+
const evalStartTime = Date.now();
|
|
544
|
+
try {
|
|
545
|
+
if (USE_MOCK_AGENT) {
|
|
546
|
+
fullTrajectory = await generateMockTrajectory(testCase);
|
|
547
|
+
for (const step of fullTrajectory) {
|
|
548
|
+
onStep(step);
|
|
549
|
+
await new Promise(r => setTimeout(r, 300));
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
else {
|
|
553
|
+
const result = await runRealAgentEvaluation(agent, modelId, testCase, onStep, onRawEvent);
|
|
554
|
+
fullTrajectory = result.trajectory;
|
|
555
|
+
agentRunId = result.runId;
|
|
556
|
+
rawEvents = result.rawEvents;
|
|
557
|
+
}
|
|
558
|
+
debug('Eval', 'Trajectory captured:', fullTrajectory.length, 'steps');
|
|
559
|
+
debug('Eval', 'Raw events captured:', rawEvents.length);
|
|
560
|
+
// TRACE MODE: Skip logs fetch and judge, return pending report
|
|
561
|
+
// Traces take ~5 minutes to propagate, so we'll poll for them later
|
|
562
|
+
if (agent.useTraces) {
|
|
563
|
+
return {
|
|
564
|
+
id: reportId,
|
|
565
|
+
timestamp: new Date().toISOString(),
|
|
566
|
+
agentName: agent.name,
|
|
567
|
+
agentKey: agent.key,
|
|
568
|
+
modelName: modelId,
|
|
569
|
+
modelId: modelId,
|
|
570
|
+
testCaseId: testCase.id,
|
|
571
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
572
|
+
status: 'completed',
|
|
573
|
+
metricsStatus: 'pending', // Will be updated after traces are available
|
|
574
|
+
trajectory: fullTrajectory,
|
|
575
|
+
metrics: {
|
|
576
|
+
accuracy: 0,
|
|
577
|
+
faithfulness: 0,
|
|
578
|
+
latency_score: 0,
|
|
579
|
+
trajectory_alignment_score: 0,
|
|
580
|
+
},
|
|
581
|
+
llmJudgeReasoning: 'Waiting for traces to become available...',
|
|
582
|
+
improvementStrategies: [],
|
|
583
|
+
runId: agentRunId || undefined,
|
|
584
|
+
sessionId: agentSessionId || undefined,
|
|
585
|
+
rawEvents,
|
|
586
|
+
};
|
|
587
|
+
}
|
|
588
|
+
// STANDARD MODE: Fetch logs and call judge immediately
|
|
589
|
+
let logs;
|
|
590
|
+
if (agentRunId) {
|
|
591
|
+
try {
|
|
592
|
+
logs = await openSearchClient.fetchLogsForRun(agentRunId);
|
|
593
|
+
}
|
|
594
|
+
catch (error) {
|
|
595
|
+
console.error('[Eval] Failed to fetch logs:', error instanceof Error ? error.message : error);
|
|
596
|
+
}
|
|
597
|
+
}
|
|
598
|
+
// Call judge
|
|
599
|
+
const models = getModels();
|
|
600
|
+
const modelConfig = models[modelId];
|
|
601
|
+
// Deprecated path — mirror the standard-mode judgeModelId resolution
|
|
602
|
+
// chain so this stays consistent if anyone still hits this code path.
|
|
603
|
+
// The agent's `modelId` is NOT used as the judge model here either.
|
|
604
|
+
const judgeModelId = process.env.BEDROCK_MODEL_ID ||
|
|
605
|
+
modelConfig?.model_id ||
|
|
606
|
+
modelId;
|
|
607
|
+
const judgment = await callBedrockJudge(fullTrajectory, {
|
|
608
|
+
expectedOutcomes: testCase.expectedOutcomes,
|
|
609
|
+
expectedTrajectory: testCase.expectedTrajectory,
|
|
610
|
+
}, logs, (chunk) => debug('Eval', 'Judge progress:', chunk.slice(0, 100)), judgeModelId, undefined, // evaluatorId not threaded on the legacy path
|
|
611
|
+
// Same forwarding as the standard branch above.
|
|
612
|
+
agentRunId || undefined);
|
|
613
|
+
debug('Eval', 'Metrics:', judgment.metrics);
|
|
614
|
+
const llmJudgeResponse = {
|
|
615
|
+
modelId: judgeModelId,
|
|
616
|
+
timestamp: new Date().toISOString(),
|
|
617
|
+
promptTokens: 0,
|
|
618
|
+
completionTokens: 0,
|
|
619
|
+
latencyMs: judgment.judgeDurationMs ?? 0,
|
|
620
|
+
// Prefer the actual unparsed model output when present (post
|
|
621
|
+
// evaluator-prompt-plumbing). Pre-fix code stuffed the parsed
|
|
622
|
+
// reasoning here — fall back to that for back-compat.
|
|
623
|
+
rawResponse: judgment.rawResponse ?? judgment.llmJudgeReasoning,
|
|
624
|
+
parsedMetrics: judgment.metrics,
|
|
625
|
+
improvementStrategies: judgment.improvementStrategies,
|
|
626
|
+
...(judgment.extraFields ? { extraFields: judgment.extraFields } : {}),
|
|
627
|
+
...(judgment.judgeDebug ? { judgeDebug: judgment.judgeDebug } : {}),
|
|
628
|
+
};
|
|
629
|
+
return {
|
|
630
|
+
id: reportId,
|
|
631
|
+
timestamp: new Date().toISOString(),
|
|
632
|
+
agentName: agent.name,
|
|
633
|
+
agentKey: agent.key,
|
|
634
|
+
modelName: modelId,
|
|
635
|
+
modelId,
|
|
636
|
+
testCaseId: testCase.id,
|
|
637
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
638
|
+
status: 'completed',
|
|
639
|
+
passFailStatus: judgment.passFailStatus,
|
|
640
|
+
trajectory: fullTrajectory,
|
|
641
|
+
metrics: judgment.metrics,
|
|
642
|
+
llmJudgeReasoning: judgment.llmJudgeReasoning,
|
|
643
|
+
// Unified judge surface (Option-B BC: legacy field above kept).
|
|
644
|
+
matcherResults: [
|
|
645
|
+
buildJudgeMatcherEntry(judgment, {
|
|
646
|
+
claim: formatExpectedOutcomesAsClaim(testCase.expectedOutcomes),
|
|
647
|
+
model: judgeModelId,
|
|
648
|
+
}),
|
|
649
|
+
],
|
|
650
|
+
improvementStrategies: judgment.improvementStrategies,
|
|
651
|
+
llmJudgeResponse,
|
|
652
|
+
openSearchLogs: logs,
|
|
653
|
+
runId: agentRunId || undefined,
|
|
654
|
+
sessionId: agentSessionId || undefined,
|
|
655
|
+
logs: logs || undefined,
|
|
656
|
+
rawEvents,
|
|
657
|
+
};
|
|
658
|
+
}
|
|
659
|
+
catch (error) {
|
|
660
|
+
console.error('[Eval] Error:', error instanceof Error ? error.message : error);
|
|
661
|
+
return {
|
|
662
|
+
id: reportId,
|
|
663
|
+
timestamp: new Date().toISOString(),
|
|
664
|
+
agentName: agent.name,
|
|
665
|
+
agentKey: agent.key,
|
|
666
|
+
modelName: modelId,
|
|
667
|
+
modelId,
|
|
668
|
+
testCaseId: testCase.id,
|
|
669
|
+
testCaseVersion: testCase.currentVersion ?? 1,
|
|
670
|
+
status: 'failed',
|
|
671
|
+
trajectory: fullTrajectory,
|
|
672
|
+
metrics: {
|
|
673
|
+
accuracy: 0,
|
|
674
|
+
faithfulness: 0,
|
|
675
|
+
latency_score: 0,
|
|
676
|
+
trajectory_alignment_score: 0,
|
|
677
|
+
},
|
|
678
|
+
llmJudgeReasoning: `Evaluation failed: ${error instanceof Error ? error.message : 'Unknown error'}`,
|
|
679
|
+
improvementStrategies: [],
|
|
680
|
+
rawEvents,
|
|
681
|
+
};
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
//# sourceMappingURL=index.js.map
|