@opensearch-project/agent-health 0.0.1 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,4548 @@
1
+ #!/usr/bin/env node
2
+
3
+ // cli/index.ts
4
+ import { Command as Command9 } from "commander";
5
+ import chalk9 from "chalk";
6
+ import { fileURLToPath as fileURLToPath3 } from "url";
7
+ import { dirname as dirname3, join as join4, resolve as resolve4 } from "path";
8
+ import { readFileSync as readFileSync3, existsSync as existsSync5 } from "fs";
9
+ import { config as loadDotenv } from "dotenv";
10
+ import open from "open";
11
+ import ora5 from "ora";
12
+
13
+ // cli/utils/startServer.ts
14
+ import { fileURLToPath } from "url";
15
+ import { dirname, join } from "path";
16
+ import { existsSync } from "fs";
17
+ var __filename = fileURLToPath(import.meta.url);
18
+ var __dirname = dirname(__filename);
19
+ function findPackageRoot() {
20
+ let dir = __dirname;
21
+ for (let i = 0; i < 5; i++) {
22
+ if (existsSync(join(dir, "package.json"))) {
23
+ return dir;
24
+ }
25
+ dir = dirname(dir);
26
+ }
27
+ return join(__dirname, "..");
28
+ }
29
+ async function startServer(options) {
30
+ process.env.VITE_BACKEND_PORT = String(options.port);
31
+ const packageRoot = findPackageRoot();
32
+ const serverPath = join(packageRoot, "server", "dist", "app.js");
33
+ const { createApp } = await import(serverPath);
34
+ const app = await createApp();
35
+ return new Promise((resolve5) => {
36
+ app.listen(options.port, "0.0.0.0", () => {
37
+ resolve5();
38
+ });
39
+ });
40
+ }
41
+
42
+ // cli/commands/list.ts
43
+ import { Command } from "commander";
44
+ import chalk from "chalk";
45
+ import Table from "cli-table3";
46
+
47
+ // lib/config/loader.ts
48
+ import { existsSync as existsSync2 } from "fs";
49
+ import { resolve } from "path";
50
+ import { pathToFileURL } from "url";
51
+
52
+ // lib/config.ts
53
+ var isServerSide = typeof window === "undefined";
54
+ var SERVER_PORT = isServerSide ? process.env?.VITE_BACKEND_PORT || process.env?.PORT || "4001" : "4001";
55
+ var BACKEND_URL = isServerSide ? `http://localhost:${SERVER_PORT}` : "";
56
+ var browserEnv = {};
57
+ if (typeof window !== "undefined" && typeof document !== "undefined") {
58
+ try {
59
+ browserEnv = import.meta?.env || {};
60
+ } catch {
61
+ }
62
+ }
63
+ var getEnvVar = (key, defaultValue) => {
64
+ if (isServerSide) {
65
+ return process.env?.[key] || defaultValue || "";
66
+ }
67
+ return browserEnv[key] || defaultValue || "";
68
+ };
69
+ var ENV_CONFIG = {
70
+ // Backend server - empty string means relative URLs
71
+ backendUrl: BACKEND_URL,
72
+ // API endpoints (derived from backend URL)
73
+ judgeApiUrl: `${BACKEND_URL}/api/judge`,
74
+ storageApiUrl: `${BACKEND_URL}/api/storage`,
75
+ agentProxyUrl: `${BACKEND_URL}/api/agent`,
76
+ openSearchProxyUrl: `${BACKEND_URL}/api/opensearch/logs`,
77
+ // AWS/Bedrock
78
+ awsRegion: getEnvVar("AWS_REGION", "us-east-1"),
79
+ awsProfile: getEnvVar("AWS_PROFILE", "default"),
80
+ bedrockModelId: getEnvVar("BEDROCK_MODEL_ID", "anthropic.claude-3-5-sonnet-20241022-v2:0"),
81
+ // OpenSearch Logs (for fetching agent observability data)
82
+ openSearchLogsEndpoint: getEnvVar("OPENSEARCH_LOGS_ENDPOINT", ""),
83
+ openSearchLogsUsername: getEnvVar("OPENSEARCH_LOGS_USERNAME", ""),
84
+ openSearchLogsPassword: getEnvVar("OPENSEARCH_LOGS_PASSWORD", ""),
85
+ openSearchLogsTracesIndex: getEnvVar("OPENSEARCH_LOGS_TRACES_INDEX", "otel-v1-apm-span-*"),
86
+ openSearchLogsIndex: getEnvVar("OPENSEARCH_LOGS_INDEX", "ml-commons-logs-*"),
87
+ // ML-Commons agent endpoint
88
+ mlcommonsEndpoint: getEnvVar("MLCOMMONS_ENDPOINT", "http://localhost:9200/_plugins/_ml/agents/{agent_id}/_execute/stream"),
89
+ // ML-Commons agent headers
90
+ mlcommonsHeaderOpenSearchUrl: getEnvVar("MLCOMMONS_HEADER_OPENSEARCH_URL", ""),
91
+ mlcommonsHeaderAuthorization: getEnvVar("MLCOMMONS_HEADER_AUTHORIZATION", ""),
92
+ mlcommonsHeaderAwsRegion: getEnvVar("MLCOMMONS_HEADER_AWS_REGION", ""),
93
+ mlcommonsHeaderAwsServiceName: getEnvVar("MLCOMMONS_HEADER_AWS_SERVICE_NAME", "es"),
94
+ mlcommonsHeaderAwsAccessKeyId: getEnvVar("MLCOMMONS_HEADER_AWS_ACCESS_KEY_ID", ""),
95
+ mlcommonsHeaderAwsSecretAccessKey: getEnvVar("MLCOMMONS_HEADER_AWS_SECRET_ACCESS_KEY", ""),
96
+ mlcommonsHeaderAwsSessionToken: getEnvVar("MLCOMMONS_HEADER_AWS_SESSION_TOKEN", ""),
97
+ // Travel Planner multi-agent endpoint (OTel Demo in Docker)
98
+ travelPlannerEndpoint: getEnvVar("TRAVEL_PLANNER_ENDPOINT", "http://localhost:3000"),
99
+ // LiteLLM (optional - for OpenAI-compatible judge/agent endpoints)
100
+ litellmApiKey: getEnvVar("LITELLM_API_KEY", ""),
101
+ litellmEndpoint: getEnvVar("LITELLM_ENDPOINT", "http://localhost:4000/v1/chat/completions"),
102
+ // Claude Code Telemetry (optional - for OTEL traces from Claude Code)
103
+ claudeCodeTelemetryEnabled: getEnvVar("CLAUDE_CODE_TELEMETRY_ENABLED", "false") === "true",
104
+ otelExporterEndpoint: getEnvVar("OTEL_EXPORTER_OTLP_ENDPOINT", ""),
105
+ otelServiceName: getEnvVar("OTEL_SERVICE_NAME", "claude-code-agent"),
106
+ otelExporterProtocol: getEnvVar("OTEL_EXPORTER_OTLP_PROTOCOL", ""),
107
+ otelExporterHeaders: getEnvVar("OTEL_EXPORTER_OTLP_HEADERS", "")
108
+ };
109
+ function buildMLCommonsHeaders() {
110
+ const headers = {};
111
+ if (ENV_CONFIG.mlcommonsHeaderOpenSearchUrl) {
112
+ headers["opensearch-url"] = ENV_CONFIG.mlcommonsHeaderOpenSearchUrl;
113
+ }
114
+ if (ENV_CONFIG.mlcommonsHeaderAwsRegion) {
115
+ headers["aws-region"] = ENV_CONFIG.mlcommonsHeaderAwsRegion;
116
+ }
117
+ if (ENV_CONFIG.mlcommonsHeaderAuthorization) {
118
+ headers["Authorization"] = ENV_CONFIG.mlcommonsHeaderAuthorization;
119
+ } else {
120
+ if (ENV_CONFIG.mlcommonsHeaderAwsServiceName) {
121
+ headers["aws-service-name"] = ENV_CONFIG.mlcommonsHeaderAwsServiceName;
122
+ }
123
+ if (ENV_CONFIG.mlcommonsHeaderAwsAccessKeyId) {
124
+ headers["aws-access-key-id"] = ENV_CONFIG.mlcommonsHeaderAwsAccessKeyId;
125
+ }
126
+ if (ENV_CONFIG.mlcommonsHeaderAwsSecretAccessKey) {
127
+ headers["aws-secret-access-key"] = ENV_CONFIG.mlcommonsHeaderAwsSecretAccessKey;
128
+ }
129
+ if (ENV_CONFIG.mlcommonsHeaderAwsSessionToken) {
130
+ headers["aws-session-token"] = ENV_CONFIG.mlcommonsHeaderAwsSessionToken;
131
+ }
132
+ }
133
+ return headers;
134
+ }
135
+
136
+ // lib/constants.ts
137
+ function getClaudeCodeConnectorEnv() {
138
+ const env = {
139
+ AWS_PROFILE: process.env.AWS_PROFILE || "Bedrock",
140
+ CLAUDE_CODE_USE_BEDROCK: "1",
141
+ AWS_REGION: process.env.AWS_REGION || "us-west-2",
142
+ DISABLE_PROMPT_CACHING: "1",
143
+ DISABLE_ERROR_REPORTING: "1"
144
+ };
145
+ if (ENV_CONFIG.claudeCodeTelemetryEnabled && ENV_CONFIG.otelExporterEndpoint) {
146
+ env.CLAUDE_CODE_ENABLE_TELEMETRY = "1";
147
+ env.OTEL_EXPORTER_OTLP_ENDPOINT = ENV_CONFIG.otelExporterEndpoint;
148
+ env.OTEL_SERVICE_NAME = ENV_CONFIG.otelServiceName;
149
+ if (ENV_CONFIG.otelExporterProtocol) {
150
+ env.OTEL_EXPORTER_OTLP_PROTOCOL = ENV_CONFIG.otelExporterProtocol;
151
+ }
152
+ if (ENV_CONFIG.otelExporterHeaders) {
153
+ env.OTEL_EXPORTER_OTLP_HEADERS = ENV_CONFIG.otelExporterHeaders;
154
+ }
155
+ } else {
156
+ env.DISABLE_TELEMETRY = "1";
157
+ }
158
+ return env;
159
+ }
160
+ var DEFAULT_CONFIG = {
161
+ agents: [
162
+ {
163
+ key: "demo",
164
+ name: "Demo Agent",
165
+ endpoint: "mock://demo",
166
+ description: "Mock agent for testing (simulated responses)",
167
+ connectorType: "mock",
168
+ models: ["demo-model"],
169
+ headers: {},
170
+ useTraces: false
171
+ },
172
+ {
173
+ key: "mlcommons-local",
174
+ name: "ML-Commons (Localhost)",
175
+ endpoint: ENV_CONFIG.mlcommonsEndpoint,
176
+ description: "Local OpenSearch ML-Commons conversational agent",
177
+ connectorType: "agui-streaming",
178
+ models: ["claude-sonnet-4.5", "claude-sonnet-4", "claude-haiku-3.5"],
179
+ headers: buildMLCommonsHeaders(),
180
+ useTraces: true
181
+ },
182
+ {
183
+ key: "travel-planner",
184
+ name: "Travel Planner",
185
+ endpoint: ENV_CONFIG.travelPlannerEndpoint,
186
+ description: "Multi-agent Travel Planner demo (requires OTel Demo running via Docker)",
187
+ connectorType: "agui-streaming",
188
+ models: ["claude-sonnet-4.5", "claude-sonnet-4", "claude-haiku-3.5"],
189
+ headers: {},
190
+ useTraces: true
191
+ },
192
+ {
193
+ key: "claude-code",
194
+ name: "Claude Code",
195
+ endpoint: "claude",
196
+ description: "Claude Code CLI agent (requires claude command installed)",
197
+ connectorType: "claude-code",
198
+ models: ["claude-sonnet-4"],
199
+ headers: {},
200
+ useTraces: ENV_CONFIG.claudeCodeTelemetryEnabled && !!ENV_CONFIG.otelExporterEndpoint,
201
+ connectorConfig: { env: getClaudeCodeConnectorEnv() }
202
+ }
203
+ ],
204
+ models: {
205
+ "demo-model": {
206
+ model_id: "mock://demo-model",
207
+ display_name: "Demo Model",
208
+ provider: "demo",
209
+ context_window: 2e5,
210
+ max_output_tokens: 4096
211
+ },
212
+ "claude-sonnet-4.5": {
213
+ model_id: "us.anthropic.claude-sonnet-4-5-20250929-v1:0",
214
+ display_name: "Claude Sonnet 4.5",
215
+ provider: "bedrock",
216
+ context_window: 2e5,
217
+ max_output_tokens: 4096
218
+ },
219
+ "claude-sonnet-4": {
220
+ model_id: "us.anthropic.claude-sonnet-4-20250514-v1:0",
221
+ display_name: "Claude Sonnet 4",
222
+ provider: "bedrock",
223
+ context_window: 2e5,
224
+ max_output_tokens: 4096
225
+ },
226
+ "claude-haiku-3.5": {
227
+ model_id: "us.anthropic.claude-3-5-haiku-20241022-v1:0",
228
+ display_name: "Claude Haiku 3.5",
229
+ provider: "bedrock",
230
+ context_window: 2e5,
231
+ max_output_tokens: 4096
232
+ },
233
+ "gpt-4o": {
234
+ model_id: "gpt-4o",
235
+ display_name: "GPT-4o (via LiteLLM)",
236
+ provider: "litellm",
237
+ context_window: 128e3,
238
+ max_output_tokens: 4096
239
+ }
240
+ },
241
+ defaults: {
242
+ retry_attempts: 2,
243
+ retry_delay_ms: 1e3
244
+ }
245
+ };
246
+
247
+ // lib/config/loader.ts
248
+ var DEFAULT_SERVER_CONFIG = {
249
+ port: 4001,
250
+ reuseExistingServer: !process.env.CI,
251
+ startTimeout: 3e4
252
+ };
253
+ var CONFIG_FILE_NAMES = [
254
+ "agent-health.config.ts",
255
+ "agent-health.config.js",
256
+ "agent-health.config.mjs"
257
+ ];
258
+ function findConfigFile(cwd = process.cwd()) {
259
+ for (const fileName of CONFIG_FILE_NAMES) {
260
+ const filePath = resolve(cwd, fileName);
261
+ if (existsSync2(filePath)) {
262
+ const format = fileName.endsWith(".ts") ? "typescript" : "javascript";
263
+ return { path: filePath, format, exists: true };
264
+ }
265
+ }
266
+ return null;
267
+ }
268
+ function toAgentConfig(userAgent) {
269
+ return {
270
+ key: userAgent.key,
271
+ name: userAgent.name,
272
+ endpoint: userAgent.endpoint,
273
+ description: userAgent.description,
274
+ enabled: userAgent.enabled ?? true,
275
+ models: userAgent.models,
276
+ headers: userAgent.headers ?? {},
277
+ useTraces: userAgent.useTraces ?? false,
278
+ connectorType: userAgent.connectorType,
279
+ connectorConfig: userAgent.connectorConfig,
280
+ hooks: userAgent.hooks
281
+ };
282
+ }
283
+ function toModelConfig(userModel) {
284
+ return [
285
+ userModel.key,
286
+ {
287
+ model_id: userModel.model_id,
288
+ display_name: userModel.display_name,
289
+ provider: userModel.provider ?? "bedrock",
290
+ context_window: userModel.context_window ?? 2e5,
291
+ max_output_tokens: userModel.max_output_tokens ?? 4096
292
+ }
293
+ ];
294
+ }
295
+ function mergeConfigs(userConfig, defaultConfig) {
296
+ const shouldExtend = userConfig.extends !== false;
297
+ let agents;
298
+ if (shouldExtend) {
299
+ const agentMap = /* @__PURE__ */ new Map();
300
+ for (const agent of defaultConfig.agents) {
301
+ agentMap.set(agent.key, agent);
302
+ }
303
+ for (const userAgent of userConfig.agents ?? []) {
304
+ agentMap.set(userAgent.key, toAgentConfig(userAgent));
305
+ }
306
+ agents = Array.from(agentMap.values());
307
+ } else {
308
+ agents = (userConfig.agents ?? []).map(toAgentConfig);
309
+ }
310
+ let models;
311
+ if (shouldExtend) {
312
+ models = { ...defaultConfig.models };
313
+ for (const userModel of userConfig.models ?? []) {
314
+ const [key, config] = toModelConfig(userModel);
315
+ models[key] = config;
316
+ }
317
+ } else {
318
+ models = {};
319
+ for (const userModel of userConfig.models ?? []) {
320
+ const [key, config] = toModelConfig(userModel);
321
+ models[key] = config;
322
+ }
323
+ }
324
+ const connectors = userConfig.connectors ?? [];
325
+ const testCases = userConfig.testCases ? Array.isArray(userConfig.testCases) ? userConfig.testCases : [userConfig.testCases] : [];
326
+ const reporters = userConfig.reporters ?? [["console"]];
327
+ const judge = userConfig.judge ?? {
328
+ provider: "bedrock",
329
+ model: "claude-sonnet-4"
330
+ };
331
+ const server = {
332
+ ...DEFAULT_SERVER_CONFIG,
333
+ ...userConfig.server
334
+ };
335
+ return {
336
+ server,
337
+ agents,
338
+ models,
339
+ connectors,
340
+ testCases,
341
+ reporters,
342
+ judge
343
+ };
344
+ }
345
+ async function loadUserConfig(configPath) {
346
+ try {
347
+ const fileUrl = pathToFileURL(configPath).href;
348
+ const module = await import(fileUrl);
349
+ return module.default ?? module;
350
+ } catch (error) {
351
+ const message = error instanceof Error ? error.message : String(error);
352
+ throw new Error(`Failed to load config file ${configPath}: ${message}`);
353
+ }
354
+ }
355
+ var cachedConfig = null;
356
+ var cachedConfigPath = null;
357
+ async function loadConfig(cwd = process.cwd(), force = false) {
358
+ const configFile = findConfigFile(cwd);
359
+ if (!force && cachedConfig && cachedConfigPath === configFile?.path) {
360
+ return cachedConfig;
361
+ }
362
+ let userConfig = {};
363
+ if (configFile) {
364
+ console.log(`[Config] Loading ${configFile.path}`);
365
+ userConfig = await loadUserConfig(configFile.path);
366
+ } else {
367
+ console.log("[Config] No config file found, using defaults + environment variables");
368
+ }
369
+ const resolved = mergeConfigs(userConfig, DEFAULT_CONFIG);
370
+ cachedConfig = resolved;
371
+ cachedConfigPath = configFile?.path ?? null;
372
+ console.log(`[Config] Loaded ${resolved.agents.length} agents, ${Object.keys(resolved.models).length} models`);
373
+ return resolved;
374
+ }
375
+ function getConfigFileInfo(cwd = process.cwd()) {
376
+ return findConfigFile(cwd);
377
+ }
378
+
379
+ // services/connectors/registry.ts
380
+ var DEFAULT_CONNECTOR_TYPE = "agui-streaming";
381
+ var ConnectorRegistryImpl = class {
382
+ constructor() {
383
+ this.connectors = /* @__PURE__ */ new Map();
384
+ }
385
+ /**
386
+ * Register a connector implementation
387
+ * @throws Error if connector with same type is already registered
388
+ */
389
+ register(connector) {
390
+ if (this.connectors.has(connector.type)) {
391
+ console.warn(
392
+ `[ConnectorRegistry] Overwriting existing connector for type: ${connector.type}`
393
+ );
394
+ }
395
+ this.connectors.set(connector.type, connector);
396
+ }
397
+ /**
398
+ * Get a connector by protocol type
399
+ */
400
+ get(type) {
401
+ return this.connectors.get(type);
402
+ }
403
+ /**
404
+ * Get all registered connectors
405
+ */
406
+ getAll() {
407
+ return Array.from(this.connectors.values());
408
+ }
409
+ /**
410
+ * Check if a connector is registered
411
+ */
412
+ has(type) {
413
+ return this.connectors.has(type);
414
+ }
415
+ /**
416
+ * Get connector for an agent config
417
+ * Handles backwards compatibility with legacy configs
418
+ *
419
+ * Resolution order:
420
+ * 1. If endpoint starts with 'mock://', use mock connector
421
+ * 2. If connectorType is specified, use that
422
+ * 3. Default to 'agui-streaming'
423
+ */
424
+ getForAgent(agent) {
425
+ if (agent.endpoint.startsWith("mock://")) {
426
+ const mockConnector2 = this.get("mock");
427
+ if (mockConnector2) {
428
+ return mockConnector2;
429
+ }
430
+ console.warn("[ConnectorRegistry] Mock connector not registered, falling back to default");
431
+ }
432
+ const connectorType = agent.connectorType ?? DEFAULT_CONNECTOR_TYPE;
433
+ const connector = this.get(connectorType);
434
+ if (!connector) {
435
+ console.error(
436
+ `[ConnectorRegistry] Connector not found for type: ${connectorType}, falling back to ${DEFAULT_CONNECTOR_TYPE}`
437
+ );
438
+ const defaultConnector = this.get(DEFAULT_CONNECTOR_TYPE);
439
+ if (!defaultConnector) {
440
+ throw new Error(
441
+ `No connector registered for type '${connectorType}' and no default connector available`
442
+ );
443
+ }
444
+ return defaultConnector;
445
+ }
446
+ return connector;
447
+ }
448
+ /**
449
+ * Clear all registered connectors (useful for testing)
450
+ */
451
+ clear() {
452
+ this.connectors.clear();
453
+ }
454
+ /**
455
+ * Get list of registered connector types
456
+ */
457
+ getRegisteredTypes() {
458
+ return Array.from(this.connectors.keys());
459
+ }
460
+ };
461
+ var connectorRegistry = new ConnectorRegistryImpl();
462
+
463
+ // lib/debug.ts
464
+ import fs from "fs";
465
+ import path from "path";
466
+ var isBrowser = typeof window !== "undefined";
467
+ var CONFIG_FILENAME = "agent-health.config.json";
468
+ var serverDebugEnabled = false;
469
+ if (!isBrowser) {
470
+ try {
471
+ const configPath = path.join(process.cwd(), CONFIG_FILENAME);
472
+ if (fs.existsSync(configPath)) {
473
+ const content = fs.readFileSync(configPath, "utf-8");
474
+ const config = JSON.parse(content) || {};
475
+ serverDebugEnabled = config.debug === true;
476
+ } else if (process.env?.DEBUG === "true") {
477
+ serverDebugEnabled = true;
478
+ }
479
+ } catch (err) {
480
+ if (process.env?.DEBUG === "true") {
481
+ serverDebugEnabled = true;
482
+ }
483
+ }
484
+ }
485
+ function isDebugEnabled() {
486
+ if (isBrowser) {
487
+ try {
488
+ return localStorage.getItem("agenteval_debug") === "true";
489
+ } catch {
490
+ return false;
491
+ }
492
+ }
493
+ return serverDebugEnabled;
494
+ }
495
+ function debug(module, ...args) {
496
+ if (isDebugEnabled()) {
497
+ console.debug(`[${module}]`, ...args);
498
+ }
499
+ }
500
+
501
+ // services/connectors/base/BaseConnector.ts
502
+ var BaseConnector = class {
503
+ /**
504
+ * Build HTTP headers from auth configuration
505
+ * @param auth Authentication configuration
506
+ * @returns Headers object ready for fetch/axios
507
+ */
508
+ buildAuthHeaders(auth) {
509
+ const headers = {};
510
+ switch (auth.type) {
511
+ case "basic":
512
+ if (auth.username && auth.password) {
513
+ const credentials = Buffer.from(`${auth.username}:${auth.password}`).toString("base64");
514
+ headers["Authorization"] = `Basic ${credentials}`;
515
+ }
516
+ break;
517
+ case "bearer":
518
+ if (auth.token) {
519
+ headers["Authorization"] = `Bearer ${auth.token}`;
520
+ }
521
+ break;
522
+ case "api-key":
523
+ if (auth.token) {
524
+ headers["X-API-Key"] = auth.token;
525
+ headers["x-api-key"] = auth.token;
526
+ }
527
+ break;
528
+ case "aws-sigv4":
529
+ console.warn("[BaseConnector] AWS SigV4 auth requires runtime signing");
530
+ break;
531
+ case "none":
532
+ default:
533
+ break;
534
+ }
535
+ if (auth.headers) {
536
+ Object.assign(headers, auth.headers);
537
+ }
538
+ return headers;
539
+ }
540
+ /**
541
+ * Build environment variables from auth configuration
542
+ * Used by subprocess connectors
543
+ */
544
+ buildAuthEnv(auth) {
545
+ const env = {};
546
+ if (auth.type === "aws-sigv4") {
547
+ if (auth.awsRegion) env["AWS_REGION"] = auth.awsRegion;
548
+ if (auth.awsAccessKeyId) env["AWS_ACCESS_KEY_ID"] = auth.awsAccessKeyId;
549
+ if (auth.awsSecretAccessKey) env["AWS_SECRET_ACCESS_KEY"] = auth.awsSecretAccessKey;
550
+ if (auth.awsSessionToken) env["AWS_SESSION_TOKEN"] = auth.awsSessionToken;
551
+ }
552
+ return env;
553
+ }
554
+ /**
555
+ * Generate a unique ID for trajectory steps
556
+ */
557
+ generateId() {
558
+ return `${Date.now()}-${Math.random().toString(36).substring(2, 9)}`;
559
+ }
560
+ /**
561
+ * Create a trajectory step with common fields
562
+ */
563
+ createStep(type, content, extra) {
564
+ return {
565
+ id: this.generateId(),
566
+ timestamp: Date.now(),
567
+ type,
568
+ content,
569
+ ...extra
570
+ };
571
+ }
572
+ /**
573
+ * Default health check implementation
574
+ * Subclasses can override for protocol-specific checks
575
+ */
576
+ async healthCheck(endpoint, auth) {
577
+ try {
578
+ const headers = this.buildAuthHeaders(auth);
579
+ const response = await fetch(endpoint, {
580
+ method: "HEAD",
581
+ headers
582
+ });
583
+ return response.ok;
584
+ } catch (error) {
585
+ console.error(`[${this.type}] Health check failed:`, error);
586
+ return false;
587
+ }
588
+ }
589
+ /**
590
+ * Log debug message with connector type prefix
591
+ */
592
+ debug(message, ...args) {
593
+ debug(this.type, message, ...args);
594
+ }
595
+ /**
596
+ * Log error message with connector type prefix
597
+ */
598
+ error(message, ...args) {
599
+ console.error(`[${this.type}] ${message}`, ...args);
600
+ }
601
+ };
602
+
603
+ // types/agui.ts
604
+ import { EventType } from "@ag-ui/core";
605
+ var AGUIEventType = EventType;
606
+
607
+ // services/agent/sseStream.ts
608
+ var SSEClient = class {
609
+ constructor() {
610
+ this.abortController = null;
611
+ }
612
+ /**
613
+ * Start consuming SSE stream from the agent endpoint
614
+ */
615
+ async consume(options) {
616
+ const {
617
+ url,
618
+ method = "POST",
619
+ headers = {},
620
+ body,
621
+ onEvent,
622
+ onError,
623
+ onComplete,
624
+ completeOnRunEnd = false,
625
+ idleTimeoutMs = 12e4
626
+ // 2 minute idle timeout by default (LLM agents can be slow)
627
+ } = options;
628
+ this.abortController = new AbortController();
629
+ debug("SSE", "Connecting to", url);
630
+ debug("SSE", "Method:", method);
631
+ debug("SSE", "Headers:", headers);
632
+ debug("SSE", "Payload:", body ? JSON.stringify(body, null, 2).substring(0, 500) : "none");
633
+ debug("SSE", "Timeout:", idleTimeoutMs, "ms");
634
+ try {
635
+ const requestConfig = {
636
+ method,
637
+ headers: {
638
+ "Content-Type": "application/json",
639
+ Accept: "text/event-stream",
640
+ ...headers
641
+ },
642
+ body: body ? JSON.stringify(body) : void 0,
643
+ signal: this.abortController.signal
644
+ };
645
+ debug("SSE", "Request config:", JSON.stringify(requestConfig, null, 2).substring(0, 500));
646
+ const response = await fetch(url, requestConfig);
647
+ debug("SSE", "Response received:", response.status, response.statusText);
648
+ if (!response.ok) {
649
+ let errorBody = "";
650
+ try {
651
+ errorBody = await response.text();
652
+ debug("SSE", "Error response body:", errorBody.substring(0, 500));
653
+ } catch {
654
+ debug("SSE", "Could not read error response body");
655
+ }
656
+ throw new Error(`HTTP ${response.status}: ${response.statusText}${errorBody ? ` - ${errorBody}` : ""}`);
657
+ }
658
+ if (!response.body) {
659
+ throw new Error("Response body is null");
660
+ }
661
+ console.info("[SSE] Connected to agent endpoint, streaming events...");
662
+ debug("SSE", "Response status:", response.status);
663
+ debug("SSE", "Content-Type:", response.headers.get("content-type"));
664
+ const completionReason = await this.processStream(response.body, onEvent, completeOnRunEnd, idleTimeoutMs);
665
+ console.info(`[SSE] Stream completed: ${completionReason}`);
666
+ debug("SSE", `Stream completed: ${completionReason}`);
667
+ onComplete?.();
668
+ } catch (error) {
669
+ if (error instanceof Error) {
670
+ if (error.name === "AbortError") {
671
+ debug("SSE", "Stream aborted (expected after run completion)");
672
+ onComplete?.();
673
+ } else {
674
+ console.error("[SSE] Stream error:", error.message);
675
+ console.error("[SSE] Debug mode is:", isDebugEnabled() ? "ENABLED \u2705" : "DISABLED \u274C");
676
+ const errorDetails = {
677
+ name: error.name,
678
+ message: error.message,
679
+ stack: error.stack,
680
+ cause: error.cause,
681
+ url,
682
+ method,
683
+ headers
684
+ };
685
+ debug("SSE", "Error details:", errorDetails);
686
+ if (error.message.includes("fetch failed")) {
687
+ debug("SSE", '\u{1F4A1} Diagnostic: "fetch failed" typically means:');
688
+ debug("SSE", " - Connection refused (endpoint not running)");
689
+ debug("SSE", " - DNS resolution failed (invalid hostname)");
690
+ debug("SSE", " - Network unreachable (firewall/VPN issues)");
691
+ debug("SSE", " - SSL/TLS certificate issues (self-signed cert)");
692
+ debug("SSE", ` Check if ${url} is accessible`);
693
+ } else if (error.message.includes("timeout")) {
694
+ debug("SSE", "\u{1F4A1} Diagnostic: Request timed out - endpoint may be slow or unresponsive");
695
+ } else if (error.message.includes("ENOTFOUND")) {
696
+ debug("SSE", "\u{1F4A1} Diagnostic: DNS lookup failed - hostname not found");
697
+ } else if (error.message.includes("ECONNREFUSED")) {
698
+ debug("SSE", "\u{1F4A1} Diagnostic: Connection refused - service not listening on this port");
699
+ }
700
+ onError?.(error);
701
+ }
702
+ } else {
703
+ console.error("[SSE] Unknown error:", error);
704
+ debug("SSE", "Unknown error details:", error);
705
+ onError?.(new Error("Unknown error occurred"));
706
+ }
707
+ }
708
+ }
709
+ /**
710
+ * Process the ReadableStream and parse SSE events
711
+ * @returns Reason for stream completion
712
+ */
713
+ async processStream(stream, onEvent, completeOnRunEnd = false, idleTimeoutMs = 12e4) {
714
+ const reader = stream.getReader();
715
+ const decoder = new TextDecoder();
716
+ let buffer = "";
717
+ let lastEventTime = Date.now();
718
+ let eventCount = 0;
719
+ let idleCheckInterval = null;
720
+ const idleTimeoutPromise = new Promise((resolve5) => {
721
+ idleCheckInterval = setInterval(() => {
722
+ const idleTime = Date.now() - lastEventTime;
723
+ if (eventCount > 0 && idleTime > idleTimeoutMs) {
724
+ debug("SSE", `Idle timeout: no events for ${idleTime}ms (threshold: ${idleTimeoutMs}ms)`);
725
+ this.abort();
726
+ resolve5("idle_timeout");
727
+ }
728
+ }, 1e3);
729
+ });
730
+ try {
731
+ const streamPromise = (async () => {
732
+ while (true) {
733
+ const { done, value } = await reader.read();
734
+ if (done) {
735
+ return "connection_closed";
736
+ }
737
+ lastEventTime = Date.now();
738
+ buffer += decoder.decode(value, { stream: true });
739
+ const lines = buffer.split("\n");
740
+ buffer = lines.pop() || "";
741
+ for (const line of lines) {
742
+ if (line.startsWith("data: ")) {
743
+ const data = line.slice(6);
744
+ if (data.trim()) {
745
+ debug("SSE", "Raw event:", data.substring(0, 200) + (data.length > 200 ? "..." : ""));
746
+ try {
747
+ const event = JSON.parse(data);
748
+ debug("SSE", "Parsed event:", event.type);
749
+ eventCount++;
750
+ if (eventCount % 10 === 0) {
751
+ console.info(`[SSE] Processed ${eventCount} events...`);
752
+ }
753
+ onEvent(event);
754
+ if (completeOnRunEnd && (event.type === AGUIEventType.RUN_FINISHED || event.type === AGUIEventType.RUN_ERROR)) {
755
+ debug("SSE", `Received ${event.type}, completing stream`);
756
+ this.abort();
757
+ return `event:${event.type}`;
758
+ }
759
+ } catch (parseError) {
760
+ console.error("[SSE] Parse error:", parseError);
761
+ debug("SSE", "Failed data:", data);
762
+ }
763
+ }
764
+ }
765
+ }
766
+ }
767
+ })();
768
+ const reason = await Promise.race([streamPromise, idleTimeoutPromise]);
769
+ return reason;
770
+ } finally {
771
+ if (idleCheckInterval) {
772
+ clearInterval(idleCheckInterval);
773
+ }
774
+ reader.releaseLock();
775
+ }
776
+ }
777
+ /**
778
+ * Abort the current stream connection
779
+ */
780
+ abort() {
781
+ this.abortController?.abort();
782
+ }
783
+ };
784
+ async function consumeSSEStream(url, payload, onEvent, headers, options) {
785
+ const client = new SSEClient();
786
+ return new Promise((resolve5, reject) => {
787
+ client.consume({
788
+ url,
789
+ method: "POST",
790
+ headers,
791
+ body: payload,
792
+ onEvent,
793
+ onError: (error) => reject(error),
794
+ onComplete: () => resolve5(),
795
+ // Enable auto-completion on RUN_FINISHED/RUN_ERROR events
796
+ // This prevents hanging when the agent doesn't close the connection
797
+ completeOnRunEnd: true,
798
+ idleTimeoutMs: options?.idleTimeoutMs
799
+ });
800
+ });
801
+ }
802
+
803
+ // services/agent/payloadBuilder.ts
804
+ var DEFAULT_PPL_TOOL = {
805
+ name: "execute_ppl_query",
806
+ description: "Update the query bar with a PPL query and optionally execute it",
807
+ parameters: {
808
+ type: "object",
809
+ properties: {
810
+ query: {
811
+ type: "string",
812
+ description: "The PPL query to set in the query bar"
813
+ },
814
+ autoExecute: {
815
+ type: "boolean",
816
+ description: "Whether to automatically execute the query (default: true)"
817
+ },
818
+ description: {
819
+ type: "string",
820
+ description: "Optional description of what the query does"
821
+ }
822
+ },
823
+ required: ["query"]
824
+ }
825
+ };
826
+ function generateId(prefix) {
827
+ const timestamp = Date.now();
828
+ const random = Math.random().toString(36).substring(2, 11);
829
+ return `${prefix}-${timestamp}-${random}`;
830
+ }
831
+ function buildAgentPayload(testCase, modelId, threadId, runId) {
832
+ const tools = testCase.tools || [DEFAULT_PPL_TOOL];
833
+ return {
834
+ threadId: threadId || generateId("thread"),
835
+ runId: runId || generateId("run"),
836
+ messages: [
837
+ {
838
+ id: generateId("msg"),
839
+ role: "user",
840
+ content: testCase.initialPrompt
841
+ }
842
+ ],
843
+ tools,
844
+ context: testCase.context || [],
845
+ state: {},
846
+ forwardedProps: {}
847
+ };
848
+ }
849
+
850
+ // services/agent/aguiConverter.ts
851
+ import { v4 as uuidv4 } from "uuid";
852
+ var AGUIToTrajectoryConverter = class {
853
+ constructor() {
854
+ this.currentTextMessage = null;
855
+ this.activeTools = /* @__PURE__ */ new Map();
856
+ this.hasEmittedAction = false;
857
+ this.runFinished = false;
858
+ this.pendingTextIsResponse = false;
859
+ this.runId = null;
860
+ this.threadId = null;
861
+ // Thinking state tracking
862
+ this.isThinking = false;
863
+ this.currentThinkingMessage = null;
864
+ }
865
+ /**
866
+ * Process a single AG UI event and convert it to TrajectoryStep(s)
867
+ * Returns an array of steps (usually 0 or 1, sometimes more)
868
+ */
869
+ processEvent(event) {
870
+ debug("Converter", `Event: ${event.type}`, JSON.stringify(event).substring(0, 300));
871
+ let steps = [];
872
+ switch (event.type) {
873
+ case AGUIEventType.RUN_STARTED:
874
+ steps = this.handleRunStarted(event);
875
+ break;
876
+ case AGUIEventType.RUN_FINISHED:
877
+ steps = this.handleRunFinished(event);
878
+ break;
879
+ case AGUIEventType.RUN_ERROR:
880
+ steps = this.handleRunError(event);
881
+ break;
882
+ case AGUIEventType.TEXT_MESSAGE_START:
883
+ steps = this.handleTextMessageStart(event);
884
+ break;
885
+ case AGUIEventType.TEXT_MESSAGE_CONTENT:
886
+ steps = this.handleTextMessageContent(event);
887
+ break;
888
+ case AGUIEventType.TEXT_MESSAGE_END:
889
+ steps = this.handleTextMessageEnd(event);
890
+ break;
891
+ case AGUIEventType.ACTIVITY_SNAPSHOT:
892
+ steps = this.handleActivitySnapshot(event);
893
+ break;
894
+ case AGUIEventType.ACTIVITY_DELTA:
895
+ steps = this.handleActivityDelta(event);
896
+ break;
897
+ case AGUIEventType.TOOL_CALL_START:
898
+ steps = this.handleToolCallStart(event);
899
+ break;
900
+ case AGUIEventType.TOOL_CALL_ARGS:
901
+ steps = this.handleToolCallArgs(event);
902
+ break;
903
+ case AGUIEventType.TOOL_CALL_END:
904
+ steps = this.handleToolCallEnd(event);
905
+ break;
906
+ case AGUIEventType.TOOL_CALL_RESULT:
907
+ steps = this.handleToolCallResult(event);
908
+ break;
909
+ // Thinking events - extended reasoning from the model
910
+ case AGUIEventType.THINKING_START:
911
+ steps = this.handleThinkingStart(event);
912
+ break;
913
+ case AGUIEventType.THINKING_END:
914
+ steps = this.handleThinkingEnd(event);
915
+ break;
916
+ case AGUIEventType.THINKING_TEXT_MESSAGE_START:
917
+ steps = this.handleThinkingTextMessageStart(event);
918
+ break;
919
+ case AGUIEventType.THINKING_TEXT_MESSAGE_CONTENT:
920
+ steps = this.handleThinkingTextMessageContent(event);
921
+ break;
922
+ case AGUIEventType.THINKING_TEXT_MESSAGE_END:
923
+ steps = this.handleThinkingTextMessageEnd(event);
924
+ break;
925
+ default:
926
+ debug("Converter", `Skipped unhandled event: ${event.type}`);
927
+ break;
928
+ }
929
+ if (steps.length > 0) {
930
+ debug("Converter", `Generated ${steps.length} step(s):`, steps.map((s) => `${s.type}${s.toolName ? `:${s.toolName}` : ""}`).join(", "));
931
+ }
932
+ return steps;
933
+ }
934
+ handleRunStarted(event) {
935
+ this.runId = event.runId;
936
+ this.threadId = event.threadId;
937
+ debug("Converter", `Run started - runId: ${this.runId}, threadId: ${this.threadId}`);
938
+ this.currentTextMessage = null;
939
+ this.activeTools.clear();
940
+ this.hasEmittedAction = false;
941
+ this.runFinished = false;
942
+ this.pendingTextIsResponse = false;
943
+ this.isThinking = false;
944
+ this.currentThinkingMessage = null;
945
+ return [];
946
+ }
947
+ getRunId() {
948
+ return this.runId;
949
+ }
950
+ getThreadId() {
951
+ return this.threadId;
952
+ }
953
+ handleRunFinished(event) {
954
+ this.runFinished = true;
955
+ if (this.currentTextMessage) {
956
+ this.pendingTextIsResponse = true;
957
+ }
958
+ return [];
959
+ }
960
+ handleRunError(event) {
961
+ console.error("[Converter] Run error:", event.message);
962
+ return [{
963
+ id: uuidv4(),
964
+ timestamp: event.timestamp,
965
+ type: "tool_result",
966
+ content: `Error: ${event.message}`,
967
+ status: "FAILURE" /* FAILURE */
968
+ }];
969
+ }
970
+ handleTextMessageStart(event) {
971
+ debug("Converter", `Text message start: ${event.messageId}`);
972
+ this.currentTextMessage = {
973
+ messageId: event.messageId,
974
+ startTime: event.timestamp,
975
+ content: ""
976
+ };
977
+ return [];
978
+ }
979
+ handleTextMessageContent(event) {
980
+ if (this.currentTextMessage && this.currentTextMessage.messageId === event.messageId) {
981
+ this.currentTextMessage.content += event.delta;
982
+ }
983
+ return [];
984
+ }
985
+ handleTextMessageEnd(event) {
986
+ if (!this.currentTextMessage || this.currentTextMessage.messageId !== event.messageId) {
987
+ debug("Converter", `Text end for unknown message: ${event.messageId}`);
988
+ return [];
989
+ }
990
+ const latencyMs = event.timestamp - this.currentTextMessage.startTime;
991
+ const content = this.currentTextMessage.content.trim();
992
+ let stepType;
993
+ if (this.pendingTextIsResponse || this.runFinished) {
994
+ stepType = "response";
995
+ } else {
996
+ stepType = "assistant";
997
+ }
998
+ debug("Converter", `Classification: ${stepType} (runFinished=${this.runFinished}, pendingResponse=${this.pendingTextIsResponse}, hasAction=${this.hasEmittedAction})`);
999
+ if (stepType === "assistant" && content.length === 0) {
1000
+ debug("Converter", "Skipping empty assistant message");
1001
+ this.currentTextMessage = null;
1002
+ return [];
1003
+ }
1004
+ const step = {
1005
+ id: uuidv4(),
1006
+ timestamp: event.timestamp,
1007
+ type: stepType,
1008
+ content,
1009
+ latencyMs
1010
+ };
1011
+ this.currentTextMessage = null;
1012
+ return [step];
1013
+ }
1014
+ handleActivitySnapshot(event) {
1015
+ const toolName = this.extractToolName(event.content.title);
1016
+ const toolArgs = this.parseToolArgs(event.content.description);
1017
+ const actionStepId = uuidv4();
1018
+ this.activeTools.set(event.messageId, {
1019
+ messageId: event.messageId,
1020
+ toolName,
1021
+ toolArgs,
1022
+ argsAccumulator: "",
1023
+ startTime: event.timestamp,
1024
+ actionStepId
1025
+ });
1026
+ this.hasEmittedAction = true;
1027
+ debug("Converter", `Tool action: ${toolName}`, toolArgs);
1028
+ return [{
1029
+ id: actionStepId,
1030
+ timestamp: event.timestamp,
1031
+ type: "action",
1032
+ content: `Calling ${toolName}...`,
1033
+ toolName,
1034
+ toolArgs
1035
+ }];
1036
+ }
1037
+ handleActivityDelta(event) {
1038
+ const toolState = this.activeTools.get(event.messageId);
1039
+ if (!toolState) return [];
1040
+ const isCompletion = event.patch.some(
1041
+ (op) => op.path === "/icon" && (op.value === "CheckCircle" || op.value === "Check")
1042
+ );
1043
+ if (!isCompletion) return [];
1044
+ const descriptionPatch = event.patch.find((op) => op.path === "/description");
1045
+ const resultContent = descriptionPatch?.value || "Tool execution completed";
1046
+ const latencyMs = event.timestamp - toolState.startTime;
1047
+ this.activeTools.delete(event.messageId);
1048
+ return [{
1049
+ id: uuidv4(),
1050
+ timestamp: event.timestamp,
1051
+ type: "tool_result",
1052
+ content: resultContent,
1053
+ status: "SUCCESS" /* SUCCESS */,
1054
+ latencyMs
1055
+ }];
1056
+ }
1057
+ handleToolCallStart(event) {
1058
+ const actionStepId = uuidv4();
1059
+ this.activeTools.set(event.toolCallId, {
1060
+ messageId: event.toolCallId,
1061
+ toolName: event.toolCallName,
1062
+ toolArgs: {},
1063
+ argsAccumulator: "",
1064
+ // Will accumulate delta strings
1065
+ startTime: event.timestamp,
1066
+ actionStepId
1067
+ });
1068
+ this.hasEmittedAction = true;
1069
+ debug("Converter", `Tool call started: ${event.toolCallName} (${event.toolCallId})`);
1070
+ return [];
1071
+ }
1072
+ handleToolCallArgs(event) {
1073
+ const toolState = this.activeTools.get(event.toolCallId);
1074
+ if (toolState) {
1075
+ toolState.argsAccumulator += event.delta;
1076
+ debug("Converter", `Tool args delta accumulated (${toolState.argsAccumulator.length} chars total)`);
1077
+ }
1078
+ return [];
1079
+ }
1080
+ /**
1081
+ * Handle TOOL_CALL_END - emits the action step with complete args
1082
+ * This is called when the agent is done sending tool call arguments
1083
+ * and expects the client to execute the tool
1084
+ */
1085
+ handleToolCallEnd(event) {
1086
+ const toolState = this.activeTools.get(event.toolCallId);
1087
+ if (!toolState) {
1088
+ debug("Converter", `Tool call end for unknown tool: ${event.toolCallId}`);
1089
+ return [];
1090
+ }
1091
+ let parsedArgs = {};
1092
+ if (toolState.argsAccumulator) {
1093
+ try {
1094
+ parsedArgs = JSON.parse(toolState.argsAccumulator);
1095
+ debug("Converter", `Tool args parsed: ${JSON.stringify(parsedArgs).substring(0, 200)}`);
1096
+ } catch {
1097
+ parsedArgs = { _raw: toolState.argsAccumulator };
1098
+ }
1099
+ }
1100
+ toolState.toolArgs = parsedArgs;
1101
+ const latencyMs = event.timestamp - toolState.startTime;
1102
+ const actionStep = {
1103
+ id: toolState.actionStepId,
1104
+ timestamp: toolState.startTime,
1105
+ type: "action",
1106
+ content: `Calling ${toolState.toolName}...`,
1107
+ toolName: toolState.toolName,
1108
+ toolArgs: parsedArgs,
1109
+ latencyMs
1110
+ };
1111
+ debug("Converter", `Tool call end: ${toolState.toolName} - action step emitted with args`);
1112
+ toolState.actionStepId = "";
1113
+ return [actionStep];
1114
+ }
1115
+ handleToolCallResult(event) {
1116
+ const toolState = this.activeTools.get(event.toolCallId);
1117
+ if (!toolState) return [];
1118
+ const latencyMs = event.timestamp - toolState.startTime;
1119
+ const steps = [];
1120
+ const actionAlreadyEmitted = !toolState.actionStepId;
1121
+ if (!actionAlreadyEmitted) {
1122
+ let parsedArgs = {};
1123
+ if (toolState.argsAccumulator) {
1124
+ try {
1125
+ parsedArgs = JSON.parse(toolState.argsAccumulator);
1126
+ debug("Converter", `Tool args parsed: ${JSON.stringify(parsedArgs).substring(0, 200)}`);
1127
+ } catch {
1128
+ parsedArgs = { _raw: toolState.argsAccumulator };
1129
+ }
1130
+ }
1131
+ toolState.toolArgs = parsedArgs;
1132
+ steps.push({
1133
+ id: toolState.actionStepId,
1134
+ timestamp: toolState.startTime,
1135
+ type: "action",
1136
+ content: `Calling ${toolState.toolName}...`,
1137
+ toolName: toolState.toolName,
1138
+ toolArgs: parsedArgs
1139
+ });
1140
+ }
1141
+ let resultContent;
1142
+ try {
1143
+ const parsed = JSON.parse(event.content);
1144
+ resultContent = typeof parsed === "string" ? parsed : JSON.stringify(parsed, null, 2);
1145
+ } catch (e) {
1146
+ resultContent = event.content;
1147
+ }
1148
+ steps.push({
1149
+ id: uuidv4(),
1150
+ timestamp: event.timestamp,
1151
+ type: "tool_result",
1152
+ content: resultContent,
1153
+ status: "SUCCESS" /* SUCCESS */,
1154
+ latencyMs
1155
+ });
1156
+ this.activeTools.delete(event.toolCallId);
1157
+ debug("Converter", `Tool call result: ${toolState.toolName} -> ${steps.length} steps emitted (action already emitted: ${actionAlreadyEmitted})`);
1158
+ return steps;
1159
+ }
1160
+ // ============ THINKING Event Handlers ============
1161
+ handleThinkingStart(event) {
1162
+ this.isThinking = true;
1163
+ debug("Converter", "Thinking started");
1164
+ return [];
1165
+ }
1166
+ handleThinkingEnd(event) {
1167
+ this.isThinking = false;
1168
+ debug("Converter", "Thinking ended");
1169
+ return [];
1170
+ }
1171
+ handleThinkingTextMessageStart(event) {
1172
+ this.currentThinkingMessage = {
1173
+ startTime: event.timestamp || Date.now(),
1174
+ content: ""
1175
+ };
1176
+ debug("Converter", "Thinking text message started");
1177
+ return [];
1178
+ }
1179
+ handleThinkingTextMessageContent(event) {
1180
+ if (this.currentThinkingMessage) {
1181
+ this.currentThinkingMessage.content += event.delta;
1182
+ }
1183
+ return [];
1184
+ }
1185
+ handleThinkingTextMessageEnd(event) {
1186
+ if (!this.currentThinkingMessage) {
1187
+ debug("Converter", "Thinking text end with no active thinking message");
1188
+ return [];
1189
+ }
1190
+ const content = this.currentThinkingMessage.content.trim();
1191
+ if (content.length === 0) {
1192
+ debug("Converter", "Skipping empty thinking message");
1193
+ this.currentThinkingMessage = null;
1194
+ return [];
1195
+ }
1196
+ const latencyMs = (event.timestamp || Date.now()) - this.currentThinkingMessage.startTime;
1197
+ const step = {
1198
+ id: uuidv4(),
1199
+ timestamp: this.currentThinkingMessage.startTime,
1200
+ type: "thinking",
1201
+ content,
1202
+ latencyMs
1203
+ };
1204
+ debug("Converter", `Thinking message completed: ${content.length} chars`);
1205
+ this.currentThinkingMessage = null;
1206
+ return [step];
1207
+ }
1208
+ extractToolName(title) {
1209
+ const runningMatch = title.match(/^Running\s+(.+)$/);
1210
+ if (runningMatch) return runningMatch[1];
1211
+ const completedMatch = title.match(/^(.+)\s+completed$/i);
1212
+ if (completedMatch) return completedMatch[1];
1213
+ return title;
1214
+ }
1215
+ parseToolArgs(description) {
1216
+ try {
1217
+ const args = {};
1218
+ const pairs = description.match(/(\w+):\s*("(?:[^"]|\\")*"|\w+)/g);
1219
+ if (pairs) {
1220
+ pairs.forEach((pair) => {
1221
+ const [key, rawValue] = pair.split(":").map((s) => s.trim());
1222
+ let value = rawValue;
1223
+ if (rawValue.startsWith('"') && rawValue.endsWith('"')) {
1224
+ value = rawValue.slice(1, -1);
1225
+ } else if (rawValue === "true") {
1226
+ value = true;
1227
+ } else if (rawValue === "false") {
1228
+ value = false;
1229
+ } else if (!isNaN(Number(rawValue))) {
1230
+ value = Number(rawValue);
1231
+ }
1232
+ args[key] = value;
1233
+ });
1234
+ return args;
1235
+ }
1236
+ return { description };
1237
+ } catch (e) {
1238
+ return { description };
1239
+ }
1240
+ }
1241
+ };
1242
+ function computeTrajectoryFromRawEvents(rawEvents) {
1243
+ const converter = new AGUIToTrajectoryConverter();
1244
+ const trajectory = [];
1245
+ for (const event of rawEvents) {
1246
+ const steps = converter.processEvent(event);
1247
+ trajectory.push(...steps);
1248
+ }
1249
+ trajectory.sort((a, b) => a.timestamp - b.timestamp);
1250
+ return trajectory;
1251
+ }
1252
+
1253
+ // services/connectors/agui/AGUIStreamingConnector.ts
1254
+ var AGUIStreamingConnector = class extends BaseConnector {
1255
+ constructor() {
1256
+ super(...arguments);
1257
+ this.type = "agui-streaming";
1258
+ this.name = "AG-UI Streaming";
1259
+ this.supportsStreaming = true;
1260
+ }
1261
+ /**
1262
+ * Build AG-UI payload from standard request
1263
+ */
1264
+ buildPayload(request) {
1265
+ return buildAgentPayload(
1266
+ request.testCase,
1267
+ request.modelId,
1268
+ request.threadId,
1269
+ request.runId
1270
+ );
1271
+ }
1272
+ /**
1273
+ * Execute the request using SSE streaming
1274
+ */
1275
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
1276
+ const hasPrebuiltPayload = !!request.payload;
1277
+ const payload = request.payload || this.buildPayload(request);
1278
+ const headers = this.buildAuthHeaders(auth);
1279
+ const trajectory = [];
1280
+ const rawEvents = [];
1281
+ const converter = new AGUIToTrajectoryConverter();
1282
+ this.debug("Executing AG-UI streaming request");
1283
+ await consumeSSEStream(
1284
+ endpoint,
1285
+ payload,
1286
+ (event) => {
1287
+ rawEvents.push(event);
1288
+ onRawEvent?.(event);
1289
+ const steps = converter.processEvent(event);
1290
+ steps.forEach((step) => {
1291
+ trajectory.push(step);
1292
+ onProgress?.(step);
1293
+ });
1294
+ },
1295
+ headers
1296
+ );
1297
+ const runId = converter.getRunId();
1298
+ this.debug("Stream completed. RunId:", runId, "Steps:", trajectory.length);
1299
+ return {
1300
+ trajectory,
1301
+ runId,
1302
+ rawEvents,
1303
+ metadata: {
1304
+ threadId: converter.getThreadId()
1305
+ }
1306
+ };
1307
+ }
1308
+ /**
1309
+ * Parse raw AG-UI events into trajectory steps
1310
+ * Used for re-processing stored raw events
1311
+ */
1312
+ parseResponse(rawEvents) {
1313
+ return computeTrajectoryFromRawEvents(rawEvents);
1314
+ }
1315
+ /**
1316
+ * Health check for AG-UI endpoint
1317
+ * Tries to connect without sending a full request
1318
+ */
1319
+ async healthCheck(endpoint, auth) {
1320
+ try {
1321
+ const headers = this.buildAuthHeaders(auth);
1322
+ const response = await fetch(endpoint, {
1323
+ method: "OPTIONS",
1324
+ headers
1325
+ });
1326
+ return true;
1327
+ } catch (error) {
1328
+ this.error("Health check failed:", error);
1329
+ return false;
1330
+ }
1331
+ }
1332
+ };
1333
+ var aguiStreamingConnector = new AGUIStreamingConnector();
1334
+
1335
+ // services/connectors/mock/MockConnector.ts
1336
+ var MockConnector = class extends BaseConnector {
1337
+ constructor() {
1338
+ super(...arguments);
1339
+ this.type = "mock";
1340
+ this.name = "Demo Agent (Mock)";
1341
+ this.supportsStreaming = true;
1342
+ }
1343
+ /**
1344
+ * Build payload - not used for mock but required by interface
1345
+ */
1346
+ buildPayload(request) {
1347
+ return {
1348
+ question: request.testCase.initialPrompt,
1349
+ context: request.testCase.context
1350
+ };
1351
+ }
1352
+ /**
1353
+ * Execute mock request - generates realistic trajectory with delays
1354
+ */
1355
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
1356
+ const trajectory = [];
1357
+ const rawEvents = [];
1358
+ const runId = `mock-run-${Date.now()}`;
1359
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
1360
+ this.debug("Generating mock trajectory for:", request.testCase.name);
1361
+ const emitStep = (step) => {
1362
+ trajectory.push(step);
1363
+ onProgress?.(step);
1364
+ rawEvents.push({ type: "MOCK_STEP", step });
1365
+ onRawEvent?.({ type: "MOCK_STEP", step });
1366
+ };
1367
+ await sleep(100);
1368
+ emitStep(this.createStep(
1369
+ "assistant",
1370
+ "I need to investigate this issue. Let me start by checking the cluster health and then drill down into specific metrics."
1371
+ ));
1372
+ await sleep(300);
1373
+ emitStep(this.createStep(
1374
+ "action",
1375
+ "Calling opensearch_cluster_health...",
1376
+ {
1377
+ toolName: "opensearch_cluster_health",
1378
+ toolArgs: { local: true }
1379
+ }
1380
+ ));
1381
+ await sleep(500);
1382
+ emitStep(this.createStep(
1383
+ "tool_result",
1384
+ JSON.stringify({ status: "yellow", number_of_nodes: 3, unassigned_shards: 0 }, null, 2),
1385
+ {
1386
+ status: "SUCCESS",
1387
+ latencyMs: 500
1388
+ }
1389
+ ));
1390
+ await sleep(300);
1391
+ emitStep(this.createStep(
1392
+ "assistant",
1393
+ "The cluster is in yellow state. Let me check the node stats to identify which node might be causing issues."
1394
+ ));
1395
+ await sleep(300);
1396
+ emitStep(this.createStep(
1397
+ "action",
1398
+ "Calling opensearch_nodes_stats...",
1399
+ {
1400
+ toolName: "opensearch_nodes_stats",
1401
+ toolArgs: { metric: "jvm,os" }
1402
+ }
1403
+ ));
1404
+ await sleep(600);
1405
+ emitStep(this.createStep(
1406
+ "tool_result",
1407
+ "Node-1: CPU 12%, JVM Heap 45%\nNode-2: CPU 15%, JVM Heap 52%\nNode-3: CPU 98%, JVM Heap 89% (Data Node)",
1408
+ {
1409
+ status: "SUCCESS",
1410
+ latencyMs: 600
1411
+ }
1412
+ ));
1413
+ await sleep(400);
1414
+ emitStep(this.createStep(
1415
+ "response",
1416
+ `## Root Cause Analysis Complete
1417
+
1418
+ **Finding:** High CPU utilization detected on Node-3 (98% CPU, 89% JVM Heap)
1419
+
1420
+ **Root Cause:** Node-3 is experiencing resource exhaustion, likely due to:
1421
+ 1. Heavy indexing or search operations
1422
+ 2. Garbage collection pressure from high heap usage
1423
+ 3. Possible hot spot in shard distribution
1424
+
1425
+ **Recommendations:**
1426
+ 1. Check hot threads on Node-3 using \`_nodes/Node-3/hot_threads\`
1427
+ 2. Review shard distribution and consider rebalancing
1428
+ 3. Monitor GC logs for long pauses
1429
+ 4. Consider scaling horizontally if load persists`
1430
+ ));
1431
+ this.debug("Mock trajectory completed. Steps:", trajectory.length);
1432
+ return {
1433
+ trajectory,
1434
+ runId,
1435
+ rawEvents,
1436
+ metadata: {
1437
+ mock: true,
1438
+ testCaseId: request.testCase.id,
1439
+ testCaseName: request.testCase.name
1440
+ }
1441
+ };
1442
+ }
1443
+ /**
1444
+ * Parse raw events - for mock, just extract steps from our custom format
1445
+ */
1446
+ parseResponse(rawEvents) {
1447
+ return rawEvents.filter((e) => e.type === "MOCK_STEP" && e.step).map((e) => e.step);
1448
+ }
1449
+ /**
1450
+ * Health check - mock is always "healthy"
1451
+ */
1452
+ async healthCheck(endpoint, auth) {
1453
+ return true;
1454
+ }
1455
+ };
1456
+ var mockConnector = new MockConnector();
1457
+
1458
+ // services/connectors/rest/RESTConnector.ts
1459
+ var RESTConnector = class extends BaseConnector {
1460
+ constructor() {
1461
+ super(...arguments);
1462
+ this.type = "rest";
1463
+ this.name = "REST API";
1464
+ this.supportsStreaming = false;
1465
+ }
1466
+ /**
1467
+ * Build generic REST payload
1468
+ * Can be customized via connectorConfig
1469
+ */
1470
+ buildPayload(request) {
1471
+ return {
1472
+ prompt: request.testCase.initialPrompt,
1473
+ context: request.testCase.context,
1474
+ model: request.modelId,
1475
+ tools: request.testCase.tools
1476
+ };
1477
+ }
1478
+ /**
1479
+ * Execute REST request
1480
+ */
1481
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
1482
+ const payload = request.payload || this.buildPayload(request);
1483
+ const headers = this.buildAuthHeaders(auth);
1484
+ this.debug("Executing REST request");
1485
+ this.debug("Endpoint:", endpoint);
1486
+ this.debug("Payload:", JSON.stringify(payload).substring(0, 500));
1487
+ const response = await fetch(endpoint, {
1488
+ method: "POST",
1489
+ headers: {
1490
+ "Content-Type": "application/json",
1491
+ ...headers
1492
+ },
1493
+ body: JSON.stringify(payload)
1494
+ });
1495
+ if (!response.ok) {
1496
+ const errorText = await response.text();
1497
+ throw new Error(`REST request failed: ${response.status} - ${errorText}`);
1498
+ }
1499
+ const data = await response.json();
1500
+ onRawEvent?.(data);
1501
+ const trajectory = this.parseResponse(data);
1502
+ trajectory.forEach((step) => onProgress?.(step));
1503
+ return {
1504
+ trajectory,
1505
+ runId: data.runId || data.id || null,
1506
+ rawEvents: [data],
1507
+ metadata: {
1508
+ status: response.status,
1509
+ responseHeaders: Object.fromEntries(response.headers.entries())
1510
+ }
1511
+ };
1512
+ }
1513
+ /**
1514
+ * Parse REST response into trajectory steps
1515
+ * This is a generic implementation - subclass for specific APIs
1516
+ */
1517
+ parseResponse(data) {
1518
+ const steps = [];
1519
+ if (data.thinking) {
1520
+ steps.push(this.createStep("thinking", data.thinking));
1521
+ }
1522
+ if (data.toolCalls && Array.isArray(data.toolCalls)) {
1523
+ for (const call of data.toolCalls) {
1524
+ steps.push(this.createStep("action", `Calling ${call.name}...`, {
1525
+ toolName: call.name,
1526
+ toolArgs: call.args || call.input
1527
+ }));
1528
+ if (call.result !== void 0) {
1529
+ steps.push(this.createStep(
1530
+ "tool_result",
1531
+ typeof call.result === "string" ? call.result : JSON.stringify(call.result),
1532
+ { status: "SUCCESS" }
1533
+ ));
1534
+ }
1535
+ }
1536
+ }
1537
+ const responseContent = data.response || data.content || data.answer || data.text || data.message;
1538
+ if (responseContent) {
1539
+ steps.push(this.createStep(
1540
+ "response",
1541
+ typeof responseContent === "string" ? responseContent : JSON.stringify(responseContent)
1542
+ ));
1543
+ }
1544
+ if (data.inference_results) {
1545
+ const outputs = data.inference_results[0]?.output || [];
1546
+ for (const output of outputs) {
1547
+ if (output.name === "response") {
1548
+ const content = output.dataAsMap?.response || output.result;
1549
+ if (content) {
1550
+ steps.push(this.createStep("response", content));
1551
+ }
1552
+ }
1553
+ }
1554
+ }
1555
+ if (steps.length === 0 && data) {
1556
+ steps.push(this.createStep("response", JSON.stringify(data, null, 2)));
1557
+ }
1558
+ return steps;
1559
+ }
1560
+ };
1561
+ var restConnector = new RESTConnector();
1562
+
1563
+ // services/connectors/litellm/LiteLLMConnector.ts
1564
+ var LiteLLMConnector = class extends BaseConnector {
1565
+ constructor() {
1566
+ super(...arguments);
1567
+ this.type = "litellm";
1568
+ this.name = "LiteLLM / OpenAI-compatible";
1569
+ this.supportsStreaming = false;
1570
+ }
1571
+ /**
1572
+ * Build OpenAI Chat Completion payload from test case
1573
+ */
1574
+ buildPayload(request) {
1575
+ const messages = [];
1576
+ if (request.testCase.context && request.testCase.context.length > 0) {
1577
+ const contextText = request.testCase.context.map((c) => typeof c === "string" ? c : JSON.stringify(c)).join("\n");
1578
+ messages.push({
1579
+ role: "system",
1580
+ content: contextText
1581
+ });
1582
+ }
1583
+ messages.push({
1584
+ role: "user",
1585
+ content: request.testCase.initialPrompt
1586
+ });
1587
+ const payload = {
1588
+ model: request.modelId,
1589
+ messages
1590
+ };
1591
+ if (request.testCase.tools && request.testCase.tools.length > 0) {
1592
+ payload.tools = request.testCase.tools.map((tool) => ({
1593
+ type: "function",
1594
+ function: {
1595
+ name: tool.name,
1596
+ description: tool.description || "",
1597
+ parameters: tool.parameters || {}
1598
+ }
1599
+ }));
1600
+ }
1601
+ return payload;
1602
+ }
1603
+ /**
1604
+ * Execute OpenAI-compatible Chat Completion request
1605
+ */
1606
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
1607
+ const payload = request.payload || this.buildPayload(request);
1608
+ const headers = this.buildAuthHeaders(auth);
1609
+ this.debug("Executing LiteLLM request");
1610
+ this.debug("Endpoint:", endpoint);
1611
+ this.debug("Model:", payload.model);
1612
+ const response = await fetch(endpoint, {
1613
+ method: "POST",
1614
+ headers: {
1615
+ "Content-Type": "application/json",
1616
+ ...headers
1617
+ },
1618
+ body: JSON.stringify(payload)
1619
+ });
1620
+ if (!response.ok) {
1621
+ const errorText = await response.text();
1622
+ throw new Error(`LiteLLM request failed: ${response.status} - ${errorText}`);
1623
+ }
1624
+ const data = await response.json();
1625
+ onRawEvent?.(data);
1626
+ const trajectory = this.parseResponse(data);
1627
+ trajectory.forEach((step) => onProgress?.(step));
1628
+ return {
1629
+ trajectory,
1630
+ runId: data.id || null,
1631
+ rawEvents: [data],
1632
+ metadata: {
1633
+ model: data.model,
1634
+ usage: data.usage,
1635
+ finishReason: data.choices?.[0]?.finish_reason
1636
+ }
1637
+ };
1638
+ }
1639
+ /**
1640
+ * Parse OpenAI Chat Completion response into trajectory steps
1641
+ */
1642
+ parseResponse(data) {
1643
+ const steps = [];
1644
+ const choice = data.choices?.[0];
1645
+ if (!choice) {
1646
+ steps.push(this.createStep("response", JSON.stringify(data, null, 2)));
1647
+ return steps;
1648
+ }
1649
+ const message = choice.message;
1650
+ if (message.tool_calls && message.tool_calls.length > 0) {
1651
+ for (const toolCall of message.tool_calls) {
1652
+ let toolArgs;
1653
+ try {
1654
+ toolArgs = JSON.parse(toolCall.function.arguments);
1655
+ } catch {
1656
+ toolArgs = toolCall.function.arguments;
1657
+ }
1658
+ steps.push(this.createStep("action", `Calling ${toolCall.function.name}...`, {
1659
+ toolName: toolCall.function.name,
1660
+ toolArgs
1661
+ }));
1662
+ }
1663
+ }
1664
+ if (message.content) {
1665
+ steps.push(this.createStep("response", message.content));
1666
+ }
1667
+ if (steps.length === 0) {
1668
+ steps.push(this.createStep("response", "(empty response)"));
1669
+ }
1670
+ return steps;
1671
+ }
1672
+ };
1673
+ var litellmConnector = new LiteLLMConnector();
1674
+
1675
+ // services/connectors/index.ts
1676
+ connectorRegistry.register(aguiStreamingConnector);
1677
+ connectorRegistry.register(mockConnector);
1678
+ connectorRegistry.register(restConnector);
1679
+ connectorRegistry.register(litellmConnector);
1680
+ console.log("[Connectors] Browser-safe connectors registered:", connectorRegistry.getRegisteredTypes().join(", "));
1681
+
1682
+ // services/connectors/subprocess/SubprocessConnector.ts
1683
+ import { spawn } from "child_process";
1684
+ var DEFAULT_SUBPROCESS_CONFIG = {
1685
+ command: "",
1686
+ args: [],
1687
+ env: {},
1688
+ inputMode: "stdin",
1689
+ outputParser: "text",
1690
+ timeout: 3e5
1691
+ // 5 minutes
1692
+ };
1693
+ var SubprocessConnector = class extends BaseConnector {
1694
+ constructor(config) {
1695
+ super();
1696
+ this.type = "subprocess";
1697
+ this.name = "Subprocess (CLI)";
1698
+ this.supportsStreaming = true;
1699
+ this.config = { ...DEFAULT_SUBPROCESS_CONFIG, ...config };
1700
+ }
1701
+ /**
1702
+ * Build input for the subprocess
1703
+ */
1704
+ buildPayload(request) {
1705
+ let prompt = request.testCase.initialPrompt;
1706
+ if (request.testCase.context && request.testCase.context.length > 0) {
1707
+ const contextStr = request.testCase.context.map((c) => `${c.description}: ${c.value}`).join("\n");
1708
+ prompt = `Context:
1709
+ ${contextStr}
1710
+
1711
+ Question: ${prompt}`;
1712
+ }
1713
+ return prompt;
1714
+ }
1715
+ /**
1716
+ * Execute subprocess and capture output
1717
+ */
1718
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
1719
+ this.debug("========== execute() STARTED ==========");
1720
+ const command = endpoint || this.config.command;
1721
+ const args = this.config.args || [];
1722
+ const input = request.payload || this.buildPayload(request);
1723
+ this.debug("Command:", command);
1724
+ this.debug("Args:", args);
1725
+ this.debug("Input mode:", this.config.inputMode);
1726
+ this.debug("Output parser:", this.config.outputParser);
1727
+ this.debug("Timeout:", this.config.timeout);
1728
+ this.debug("Input (first 500 chars):", input.substring(0, 500));
1729
+ this.debug("Working dir:", this.config.workingDir || process.cwd());
1730
+ const env = {
1731
+ ...process.env,
1732
+ ...this.buildAuthEnv(auth),
1733
+ ...this.config.env
1734
+ };
1735
+ return new Promise((resolve5, reject) => {
1736
+ const trajectory = [];
1737
+ const rawOutput = [];
1738
+ let stdout = "";
1739
+ let stderr = "";
1740
+ let settled = false;
1741
+ const finalArgs = this.config.inputMode === "arg" ? [...args, input] : args;
1742
+ this.debug("Spawning process...");
1743
+ this.debug("Full command:", command, finalArgs.join(" "));
1744
+ const proc = spawn(command, finalArgs, {
1745
+ env,
1746
+ cwd: this.config.workingDir,
1747
+ shell: true
1748
+ });
1749
+ this.debug("Process spawned, PID:", proc.pid);
1750
+ const timeoutId = setTimeout(() => {
1751
+ if (settled) return;
1752
+ settled = true;
1753
+ this.debug("TIMEOUT reached, killing process");
1754
+ proc.kill("SIGTERM");
1755
+ reject(new Error(`Subprocess timed out after ${this.config.timeout}ms`));
1756
+ }, this.config.timeout);
1757
+ if (this.config.inputMode === "stdin") {
1758
+ this.debug("Writing input to stdin...");
1759
+ proc.stdin.write(input);
1760
+ proc.stdin.end();
1761
+ this.debug("stdin closed");
1762
+ }
1763
+ proc.stdout.on("data", (data) => {
1764
+ const chunk = data.toString();
1765
+ this.debug("stdout received:", chunk.length, "bytes");
1766
+ this.debug("stdout preview:", chunk.substring(0, 200));
1767
+ stdout += chunk;
1768
+ rawOutput.push({ type: "stdout", data: chunk, timestamp: Date.now() });
1769
+ onRawEvent?.({ type: "stdout", data: chunk });
1770
+ if (this.config.outputParser === "streaming") {
1771
+ this.parseStreamingOutput(chunk, trajectory, onProgress);
1772
+ }
1773
+ });
1774
+ proc.stderr.on("data", (data) => {
1775
+ const chunk = data.toString();
1776
+ this.debug("stderr received:", chunk.length, "bytes");
1777
+ this.debug("stderr:", chunk);
1778
+ stderr += chunk;
1779
+ onRawEvent?.({ type: "stderr", data: chunk });
1780
+ this.debug("stderr:", chunk);
1781
+ });
1782
+ proc.on("close", (code, signal) => {
1783
+ this.debug("Process closed with code:", code, "signal:", signal);
1784
+ clearTimeout(timeoutId);
1785
+ if (settled) return;
1786
+ settled = true;
1787
+ if (code !== 0) {
1788
+ this.debug("Non-zero exit code:", code);
1789
+ this.error(`Process exited with code ${code}`);
1790
+ this.error("stderr:", stderr);
1791
+ }
1792
+ const finalTrajectory = this.config.outputParser === "streaming" ? trajectory : this.parseResponse({ stdout, stderr, exitCode: code });
1793
+ if (this.config.outputParser !== "streaming") {
1794
+ finalTrajectory.forEach((step) => onProgress?.(step));
1795
+ }
1796
+ this.debug("Resolving with trajectory of", finalTrajectory.length, "steps");
1797
+ resolve5({
1798
+ trajectory: finalTrajectory,
1799
+ runId: `subprocess-${Date.now()}`,
1800
+ rawEvents: rawOutput,
1801
+ metadata: {
1802
+ command,
1803
+ args: finalArgs,
1804
+ exitCode: code,
1805
+ stderr: stderr || void 0
1806
+ }
1807
+ });
1808
+ });
1809
+ proc.on("error", (error) => {
1810
+ this.debug("ERROR event:", error.message);
1811
+ clearTimeout(timeoutId);
1812
+ if (settled) return;
1813
+ settled = true;
1814
+ let errorMsg = `Failed to spawn subprocess: ${error.message}`;
1815
+ if (error.message.includes("ENOENT")) {
1816
+ errorMsg = `Command '${command}' not found. Is it installed and in PATH?`;
1817
+ console.error(`[Subprocess] ENOENT error - command '${command}' not found in PATH`);
1818
+ } else if (error.message.includes("EACCES")) {
1819
+ errorMsg = `Permission denied executing '${command}'. Check file permissions.`;
1820
+ console.error(`[Subprocess] EACCES error - permission denied for '${command}'`);
1821
+ } else if (error.message.includes("EPERM")) {
1822
+ errorMsg = `Operation not permitted for '${command}'. May require elevated privileges.`;
1823
+ console.error(`[Subprocess] EPERM error - operation not permitted`);
1824
+ }
1825
+ reject(new Error(errorMsg));
1826
+ });
1827
+ });
1828
+ this.debug("========== execute() COMPLETED ==========");
1829
+ }
1830
+ /**
1831
+ * Parse streaming output and emit steps in real-time
1832
+ */
1833
+ parseStreamingOutput(chunk, trajectory, onProgress) {
1834
+ const lines = chunk.split("\n").filter((line) => line.trim());
1835
+ for (const line of lines) {
1836
+ const step = this.createStep("assistant", line);
1837
+ trajectory.push(step);
1838
+ onProgress?.(step);
1839
+ }
1840
+ }
1841
+ /**
1842
+ * Parse final subprocess output
1843
+ */
1844
+ parseResponse(data) {
1845
+ const steps = [];
1846
+ if (this.config.outputParser === "json") {
1847
+ try {
1848
+ const parsed = JSON.parse(data.stdout);
1849
+ return this.parseJsonOutput(parsed);
1850
+ } catch {
1851
+ this.debug("Failed to parse JSON output, falling back to text");
1852
+ }
1853
+ }
1854
+ if (data.stdout.trim()) {
1855
+ steps.push(this.createStep("response", data.stdout.trim()));
1856
+ }
1857
+ if (data.exitCode !== 0 && data.stderr.trim()) {
1858
+ steps.push(this.createStep("tool_result", `Error: ${data.stderr.trim()}`, {
1859
+ status: "FAILURE"
1860
+ }));
1861
+ }
1862
+ return steps;
1863
+ }
1864
+ /**
1865
+ * Parse JSON output into trajectory steps
1866
+ */
1867
+ parseJsonOutput(data) {
1868
+ const steps = [];
1869
+ if (data.thinking) {
1870
+ steps.push(this.createStep("thinking", data.thinking));
1871
+ }
1872
+ if (data.steps && Array.isArray(data.steps)) {
1873
+ for (const step of data.steps) {
1874
+ steps.push(this.createStep(step.type || "assistant", step.content, {
1875
+ toolName: step.toolName,
1876
+ toolArgs: step.toolArgs
1877
+ }));
1878
+ }
1879
+ }
1880
+ if (data.response || data.answer || data.content) {
1881
+ steps.push(this.createStep("response", data.response || data.answer || data.content));
1882
+ }
1883
+ return steps;
1884
+ }
1885
+ /**
1886
+ * Health check - verify command exists
1887
+ */
1888
+ async healthCheck(endpoint, auth) {
1889
+ const command = endpoint || this.config.command;
1890
+ if (!command) return false;
1891
+ return new Promise((resolve5) => {
1892
+ const proc = spawn("which", [command], { shell: true });
1893
+ proc.on("close", (code) => resolve5(code === 0));
1894
+ proc.on("error", () => resolve5(false));
1895
+ });
1896
+ }
1897
+ };
1898
+ var subprocessConnector = new SubprocessConnector();
1899
+
1900
+ // services/connectors/claude-code/ClaudeCodeConnector.ts
1901
+ var CLAUDE_CODE_DEFAULT_CONFIG = {
1902
+ command: "claude",
1903
+ args: ["--print", "--verbose", "--output-format", "stream-json"],
1904
+ // Structured JSON output (--verbose required with stream-json)
1905
+ env: {
1906
+ // These can be overridden by agent config or environment
1907
+ DISABLE_PROMPT_CACHING: "1",
1908
+ DISABLE_ERROR_REPORTING: "1"
1909
+ // Note: DISABLE_TELEMETRY removed - telemetry enabled by default
1910
+ // Configure OTEL_EXPORTER_OTLP_ENDPOINT in .env to send traces
1911
+ },
1912
+ inputMode: "stdin",
1913
+ outputParser: "streaming",
1914
+ timeout: 6e5
1915
+ // 10 minutes for Claude Code
1916
+ };
1917
+ var ClaudeCodeConnector = class extends SubprocessConnector {
1918
+ constructor(config) {
1919
+ super({ ...CLAUDE_CODE_DEFAULT_CONFIG, ...config });
1920
+ this.type = "claude-code";
1921
+ this.name = "Claude Code CLI";
1922
+ this.outputBuffer = "";
1923
+ this.thinkingBuffer = "";
1924
+ this.isInThinking = false;
1925
+ }
1926
+ /**
1927
+ * Build prompt for Claude Code
1928
+ * Structures the input to get the best RCA results
1929
+ */
1930
+ buildPayload(request) {
1931
+ const parts = [];
1932
+ if (request.testCase.context && request.testCase.context.length > 0) {
1933
+ parts.push("## Context");
1934
+ for (const ctx of request.testCase.context) {
1935
+ parts.push(`**${ctx.description}:**`);
1936
+ parts.push(ctx.value);
1937
+ parts.push("");
1938
+ }
1939
+ }
1940
+ parts.push("## Task");
1941
+ parts.push(request.testCase.initialPrompt);
1942
+ return parts.join("\n");
1943
+ }
1944
+ /**
1945
+ * Parse Claude Code streaming output (stream-json format)
1946
+ * Each line is a JSON object with type and content
1947
+ */
1948
+ parseStreamingOutput(chunk, trajectory, onProgress) {
1949
+ this.outputBuffer += chunk;
1950
+ const lines = this.outputBuffer.split("\n");
1951
+ this.outputBuffer = lines.pop() || "";
1952
+ for (const line of lines) {
1953
+ const trimmed = line.trim();
1954
+ if (!trimmed) continue;
1955
+ try {
1956
+ const event = JSON.parse(trimmed);
1957
+ const steps = this.parseJsonEvent(event);
1958
+ for (const step of steps) {
1959
+ trajectory.push(step);
1960
+ onProgress?.(step);
1961
+ }
1962
+ } catch {
1963
+ if (trimmed) {
1964
+ const step = this.createStep("assistant", trimmed);
1965
+ trajectory.push(step);
1966
+ onProgress?.(step);
1967
+ }
1968
+ }
1969
+ }
1970
+ }
1971
+ /**
1972
+ * Parse a single JSON event from stream-json output
1973
+ */
1974
+ parseJsonEvent(event) {
1975
+ const steps = [];
1976
+ if (event.type === "assistant" && event.message?.content) {
1977
+ for (const block of event.message.content) {
1978
+ if (block.type === "thinking" && block.thinking) {
1979
+ steps.push(this.createStep("thinking", block.thinking));
1980
+ } else if (block.type === "text" && block.text) {
1981
+ steps.push(this.createStep("assistant", block.text));
1982
+ } else if (block.type === "tool_use") {
1983
+ steps.push(this.createStep("action", JSON.stringify(block.input || {}), {
1984
+ toolName: block.name,
1985
+ toolArgs: block.input
1986
+ }));
1987
+ }
1988
+ }
1989
+ } else if (event.type === "content_block_delta") {
1990
+ if (event.delta?.type === "thinking_delta" && event.delta.thinking) {
1991
+ this.thinkingBuffer += event.delta.thinking;
1992
+ } else if (event.delta?.type === "text_delta" && event.delta.text) {
1993
+ steps.push(this.createStep("assistant", event.delta.text));
1994
+ }
1995
+ } else if (event.type === "content_block_stop" && this.thinkingBuffer) {
1996
+ steps.push(this.createStep("thinking", this.thinkingBuffer));
1997
+ this.thinkingBuffer = "";
1998
+ } else if (event.type === "result" && event.result) {
1999
+ steps.push(this.createStep(
2000
+ "response",
2001
+ typeof event.result === "string" ? event.result : JSON.stringify(event.result)
2002
+ ));
2003
+ } else if (event.type === "tool_result") {
2004
+ steps.push(this.createStep(
2005
+ "tool_result",
2006
+ typeof event.content === "string" ? event.content : JSON.stringify(event.content),
2007
+ { status: event.is_error ? "FAILURE" /* FAILURE */ : "SUCCESS" /* SUCCESS */ }
2008
+ ));
2009
+ }
2010
+ return steps;
2011
+ }
2012
+ /**
2013
+ * Parse final output for Claude Code
2014
+ */
2015
+ parseResponse(data) {
2016
+ const steps = [];
2017
+ let content = data.stdout;
2018
+ const thinkingMatches = content.matchAll(/<thinking>([\s\S]*?)<\/thinking>/g);
2019
+ for (const match of thinkingMatches) {
2020
+ const thinking = match[1].trim();
2021
+ if (thinking) {
2022
+ steps.push(this.createStep("thinking", thinking));
2023
+ }
2024
+ content = content.replace(match[0], "");
2025
+ }
2026
+ const response = content.trim();
2027
+ if (response) {
2028
+ steps.push(this.createStep("response", response));
2029
+ }
2030
+ if (data.exitCode !== 0 && data.stderr.trim()) {
2031
+ steps.push(this.createStep("tool_result", `Error: ${data.stderr.trim()}`, {
2032
+ status: "FAILURE" /* FAILURE */
2033
+ }));
2034
+ }
2035
+ return steps;
2036
+ }
2037
+ /**
2038
+ * Reset state for new execution
2039
+ */
2040
+ resetState() {
2041
+ this.outputBuffer = "";
2042
+ this.thinkingBuffer = "";
2043
+ this.isInThinking = false;
2044
+ }
2045
+ /**
2046
+ * Override execute to reset state
2047
+ */
2048
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
2049
+ this.debug("========== execute() STARTED ==========");
2050
+ this.debug("Endpoint:", endpoint);
2051
+ this.debug("Test case:", request.testCase.name);
2052
+ this.debug("Config:", this["config"]);
2053
+ this.resetState();
2054
+ this.debug("State reset, calling super.execute()...");
2055
+ const result = await super.execute(endpoint, request, auth, onProgress, onRawEvent);
2056
+ this.debug("super.execute() returned with", result.trajectory.length, "steps");
2057
+ this.debug("========== execute() COMPLETED ==========");
2058
+ return result;
2059
+ }
2060
+ /**
2061
+ * Health check - verify claude command exists
2062
+ */
2063
+ async healthCheck(endpoint, auth) {
2064
+ return super.healthCheck(endpoint || "claude", auth);
2065
+ }
2066
+ };
2067
+ var claudeCodeConnector = new ClaudeCodeConnector();
2068
+
2069
+ // services/connectors/server.ts
2070
+ connectorRegistry.register(subprocessConnector);
2071
+ connectorRegistry.register(claudeCodeConnector);
2072
+ console.log("[Connectors] Server connectors registered:", connectorRegistry.getRegisteredTypes().join(", "));
2073
+
2074
+ // cli/utils/serverLifecycle.ts
2075
+ import { spawn as spawn2, execSync } from "child_process";
2076
+ import net from "net";
2077
+ import { readFileSync } from "fs";
2078
+ import { fileURLToPath as fileURLToPath2 } from "url";
2079
+ import { dirname as dirname2, join as join3 } from "path";
2080
+ var __filename2 = fileURLToPath2(import.meta.url);
2081
+ var __dirname2 = dirname2(__filename2);
2082
+ var packageJsonPath = join3(__dirname2, "..", "..", "package.json");
2083
+ var cachedVersion = null;
2084
+ function getCliVersion() {
2085
+ if (cachedVersion !== null) {
2086
+ return cachedVersion;
2087
+ }
2088
+ try {
2089
+ const packageJson = JSON.parse(readFileSync(packageJsonPath, "utf-8"));
2090
+ cachedVersion = packageJson.version || "unknown";
2091
+ } catch {
2092
+ try {
2093
+ const altPath = join3(__dirname2, "..", "..", "..", "package.json");
2094
+ const packageJson = JSON.parse(readFileSync(altPath, "utf-8"));
2095
+ cachedVersion = packageJson.version || "unknown";
2096
+ } catch {
2097
+ cachedVersion = "unknown";
2098
+ }
2099
+ }
2100
+ return cachedVersion;
2101
+ }
2102
+ async function isServerRunning(port) {
2103
+ const controller = new AbortController();
2104
+ const timeout = setTimeout(() => controller.abort(), 2e3);
2105
+ try {
2106
+ const response = await fetch(`http://localhost:${port}/health`, {
2107
+ signal: controller.signal
2108
+ });
2109
+ if (response.ok) {
2110
+ return true;
2111
+ }
2112
+ } catch {
2113
+ } finally {
2114
+ clearTimeout(timeout);
2115
+ }
2116
+ return new Promise((resolve5) => {
2117
+ const socket = new net.Socket();
2118
+ socket.setTimeout(1e3);
2119
+ socket.on("connect", () => {
2120
+ socket.destroy();
2121
+ resolve5(true);
2122
+ });
2123
+ socket.on("timeout", () => {
2124
+ socket.destroy();
2125
+ resolve5(false);
2126
+ });
2127
+ socket.on("error", () => {
2128
+ resolve5(false);
2129
+ });
2130
+ socket.connect(port, "localhost");
2131
+ });
2132
+ }
2133
+ async function checkServerStatus(port) {
2134
+ const controller = new AbortController();
2135
+ const timeout = setTimeout(() => controller.abort(), 2e3);
2136
+ try {
2137
+ const response = await fetch(`http://localhost:${port}/health`, {
2138
+ signal: controller.signal
2139
+ });
2140
+ if (response.ok) {
2141
+ const data = await response.json();
2142
+ return {
2143
+ running: true,
2144
+ version: data.version
2145
+ };
2146
+ }
2147
+ } catch {
2148
+ } finally {
2149
+ clearTimeout(timeout);
2150
+ }
2151
+ return { running: false };
2152
+ }
2153
+ async function killServerOnPort(port) {
2154
+ try {
2155
+ if (process.platform !== "win32") {
2156
+ try {
2157
+ execSync(`lsof -t -i:${port} -sTCP:LISTEN | xargs kill -9 2>/dev/null || true`, { stdio: "ignore" });
2158
+ } catch {
2159
+ }
2160
+ } else {
2161
+ try {
2162
+ const result = execSync(`netstat -ano | findstr :${port}`, { encoding: "utf-8" });
2163
+ const lines = result.trim().split("\n");
2164
+ for (const line of lines) {
2165
+ const parts = line.trim().split(/\s+/);
2166
+ const pid = parts[parts.length - 1];
2167
+ if (pid && !isNaN(parseInt(pid))) {
2168
+ try {
2169
+ execSync(`taskkill /PID ${pid} /F`, { stdio: "ignore" });
2170
+ } catch {
2171
+ }
2172
+ }
2173
+ }
2174
+ } catch {
2175
+ }
2176
+ }
2177
+ const maxRetries = 10;
2178
+ const retryDelay = 500;
2179
+ for (let i = 0; i < maxRetries; i++) {
2180
+ await new Promise((r) => setTimeout(r, retryDelay));
2181
+ const stillRunning = await isServerRunning(port);
2182
+ if (!stillRunning) {
2183
+ return;
2184
+ }
2185
+ }
2186
+ console.warn(`[ServerLifecycle] Port ${port} may still be in use after ${maxRetries} retries`);
2187
+ } catch {
2188
+ }
2189
+ }
2190
+ async function waitForServer(port, timeout) {
2191
+ const startTime = Date.now();
2192
+ const pollInterval = 500;
2193
+ while (Date.now() - startTime < timeout) {
2194
+ if (await isServerRunning(port)) {
2195
+ return true;
2196
+ }
2197
+ await new Promise((r) => setTimeout(r, pollInterval));
2198
+ }
2199
+ return false;
2200
+ }
2201
+ async function startServer2(port, timeout) {
2202
+ const packageRoot = join3(__dirname2, "..", "..");
2203
+ const cliPath = join3(packageRoot, "bin", "cli.js");
2204
+ const child = spawn2("node", [cliPath, "serve", "-p", String(port), "--no-browser"], {
2205
+ detached: true,
2206
+ stdio: ["ignore", "pipe", "pipe"],
2207
+ env: {
2208
+ ...process.env
2209
+ }
2210
+ });
2211
+ let stderrOutput = "";
2212
+ let stdoutOutput = "";
2213
+ child.stderr?.on("data", (data) => {
2214
+ stderrOutput += data.toString();
2215
+ });
2216
+ child.stdout?.on("data", (data) => {
2217
+ stdoutOutput += data.toString();
2218
+ });
2219
+ let earlyExit = false;
2220
+ let exitCode = null;
2221
+ child.on("exit", (code) => {
2222
+ earlyExit = true;
2223
+ exitCode = code;
2224
+ });
2225
+ child.unref();
2226
+ const ready = await waitForServer(port, timeout);
2227
+ if (!ready) {
2228
+ try {
2229
+ child.kill();
2230
+ } catch {
2231
+ }
2232
+ if (earlyExit) {
2233
+ console.error(`[ServerLifecycle] Server process exited with code ${exitCode} before becoming ready`);
2234
+ } else {
2235
+ console.error(`[ServerLifecycle] Server process did not respond to health checks within ${timeout}ms`);
2236
+ }
2237
+ if (stderrOutput) {
2238
+ console.error(`[ServerLifecycle] Server stderr:
2239
+ ${stderrOutput}`);
2240
+ }
2241
+ if (stdoutOutput) {
2242
+ console.error(`[ServerLifecycle] Server stdout:
2243
+ ${stdoutOutput}`);
2244
+ }
2245
+ if (!stderrOutput && !stdoutOutput) {
2246
+ console.error(`[ServerLifecycle] No output captured from server process`);
2247
+ console.error(`[ServerLifecycle] CLI path: ${cliPath}`);
2248
+ console.error(`[ServerLifecycle] Package root: ${packageRoot}`);
2249
+ }
2250
+ throw new Error(`Server failed to start within ${timeout}ms on port ${port}`);
2251
+ }
2252
+ return child;
2253
+ }
2254
+ function stopServer(process2) {
2255
+ try {
2256
+ if (process2.pid) {
2257
+ try {
2258
+ process2.kill("SIGTERM");
2259
+ } catch {
2260
+ }
2261
+ }
2262
+ } catch {
2263
+ }
2264
+ }
2265
+ async function ensureServer(config) {
2266
+ const { port, reuseExistingServer, startTimeout } = config;
2267
+ const baseUrl = `http://localhost:${port}`;
2268
+ const serverStatus = await checkServerStatus(port);
2269
+ const cliVersion = getCliVersion();
2270
+ if (serverStatus.running) {
2271
+ const versionMatches = serverStatus.version === cliVersion || serverStatus.version === "unknown" || cliVersion === "unknown";
2272
+ if (!versionMatches) {
2273
+ console.log(`[ServerLifecycle] Version mismatch detected!`);
2274
+ console.log(`[ServerLifecycle] Server version: ${serverStatus.version}`);
2275
+ console.log(`[ServerLifecycle] CLI version: ${cliVersion}`);
2276
+ if (reuseExistingServer) {
2277
+ console.log(`[ServerLifecycle] Stopping old server and starting v${cliVersion}...`);
2278
+ await killServerOnPort(port);
2279
+ } else {
2280
+ throw new Error(
2281
+ `Server version mismatch: server=${serverStatus.version}, CLI=${cliVersion}. Stop the existing server or upgrade to matching version.`
2282
+ );
2283
+ }
2284
+ } else if (reuseExistingServer) {
2285
+ console.log(`[ServerLifecycle] Reusing existing server (version ${serverStatus.version})`);
2286
+ return {
2287
+ wasStarted: false,
2288
+ baseUrl
2289
+ };
2290
+ } else {
2291
+ throw new Error(
2292
+ `Server already running on port ${port}. In CI mode (reuseExistingServer=false), this is an error. Stop the existing server or set reuseExistingServer: true.`
2293
+ );
2294
+ }
2295
+ }
2296
+ const serverProcess = await startServer2(port, startTimeout);
2297
+ return {
2298
+ wasStarted: true,
2299
+ baseUrl,
2300
+ process: serverProcess
2301
+ };
2302
+ }
2303
+ function createServerCleanup(result, isCI) {
2304
+ return () => {
2305
+ if (result.wasStarted && isCI && result.process) {
2306
+ stopServer(result.process);
2307
+ }
2308
+ };
2309
+ }
2310
+
2311
+ // cli/utils/apiClient.ts
2312
+ var ApiClient = class {
2313
+ constructor(baseUrl) {
2314
+ this.baseUrl = baseUrl;
2315
+ }
2316
+ /**
2317
+ * Check if server is healthy, with optional retries and exponential backoff.
2318
+ */
2319
+ async checkHealth(retries = 2, delayMs = 500) {
2320
+ let lastError;
2321
+ for (let attempt = 0; attempt <= retries; attempt++) {
2322
+ try {
2323
+ const res = await fetch(`${this.baseUrl}/health`);
2324
+ if (!res.ok) {
2325
+ throw new Error(`Server health check failed: ${res.status} ${res.statusText}`);
2326
+ }
2327
+ return res.json();
2328
+ } catch (err) {
2329
+ lastError = err instanceof Error ? err : new Error(String(err));
2330
+ if (attempt < retries) {
2331
+ await new Promise((resolve5) => setTimeout(resolve5, delayMs * Math.pow(2, attempt)));
2332
+ }
2333
+ }
2334
+ }
2335
+ throw lastError;
2336
+ }
2337
+ /**
2338
+ * List all benchmarks
2339
+ */
2340
+ async listBenchmarks() {
2341
+ const res = await fetch(`${this.baseUrl}/api/storage/benchmarks`);
2342
+ if (!res.ok) {
2343
+ throw new Error(`Failed to list benchmarks: ${res.status} ${res.statusText}`);
2344
+ }
2345
+ const data = await res.json();
2346
+ return data.benchmarks || [];
2347
+ }
2348
+ /**
2349
+ * Get benchmark by ID
2350
+ */
2351
+ async getBenchmark(id) {
2352
+ const res = await fetch(`${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(id)}`);
2353
+ if (res.status === 404) {
2354
+ return null;
2355
+ }
2356
+ if (!res.ok) {
2357
+ throw new Error(`Failed to get benchmark: ${res.status} ${res.statusText}`);
2358
+ }
2359
+ return res.json();
2360
+ }
2361
+ /**
2362
+ * Find benchmark by name or ID
2363
+ *
2364
+ * Prioritizes:
2365
+ * 1. Exact ID match
2366
+ * 2. Exact name match (case-sensitive)
2367
+ */
2368
+ async findBenchmark(identifier) {
2369
+ const byId = await this.getBenchmark(identifier);
2370
+ if (byId) {
2371
+ return byId;
2372
+ }
2373
+ const benchmarks = await this.listBenchmarks();
2374
+ return benchmarks.find((b) => b.name === identifier) || null;
2375
+ }
2376
+ /**
2377
+ * Execute benchmark run (SSE stream)
2378
+ *
2379
+ * Streams progress events and returns the completed run.
2380
+ * If the SSE stream disconnects, falls back to polling for status.
2381
+ */
2382
+ async executeBenchmark(benchmarkId, runConfig, onProgress) {
2383
+ const res = await fetch(
2384
+ `${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmarkId)}/execute`,
2385
+ {
2386
+ method: "POST",
2387
+ headers: { "Content-Type": "application/json" },
2388
+ body: JSON.stringify(runConfig)
2389
+ }
2390
+ );
2391
+ if (!res.ok) {
2392
+ const errorBody = await res.text();
2393
+ let errorMessage;
2394
+ try {
2395
+ const parsed = JSON.parse(errorBody);
2396
+ errorMessage = parsed.error || errorBody;
2397
+ } catch {
2398
+ errorMessage = errorBody;
2399
+ }
2400
+ throw new Error(`Failed to execute benchmark: ${errorMessage}`);
2401
+ }
2402
+ if (!res.body) {
2403
+ throw new Error("Response body is missing");
2404
+ }
2405
+ const reader = res.body.getReader();
2406
+ const decoder = new TextDecoder();
2407
+ let buffer = "";
2408
+ let finalRun = null;
2409
+ let runId = null;
2410
+ try {
2411
+ while (true) {
2412
+ const { done, value } = await reader.read();
2413
+ if (done) break;
2414
+ buffer += decoder.decode(value, { stream: true });
2415
+ const lines = buffer.split("\n\n");
2416
+ buffer = lines.pop() || "";
2417
+ for (const line of lines) {
2418
+ if (line.startsWith("data: ")) {
2419
+ try {
2420
+ const event = JSON.parse(line.slice(6));
2421
+ onProgress?.(event);
2422
+ if (event.type === "started") {
2423
+ runId = event.runId;
2424
+ } else if (event.type === "completed" || event.type === "cancelled") {
2425
+ finalRun = event.run;
2426
+ } else if (event.type === "error") {
2427
+ throw new Error(event.error);
2428
+ }
2429
+ } catch (e) {
2430
+ if (e instanceof SyntaxError) continue;
2431
+ throw e;
2432
+ }
2433
+ }
2434
+ }
2435
+ }
2436
+ } catch (streamError) {
2437
+ if (runId) {
2438
+ console.warn(`[ApiClient] SSE stream disconnected: ${streamError instanceof Error ? streamError.message : streamError}`);
2439
+ console.warn(`[ApiClient] Falling back to polling for run ${runId}...`);
2440
+ await new Promise((resolve5) => setTimeout(resolve5, 2e3));
2441
+ const polledRun = await this.pollRunStatus(benchmarkId, runId, (run) => {
2442
+ const completedCount = Object.values(run.results || {}).filter(
2443
+ (r) => r.status === "completed" || r.status === "failed"
2444
+ ).length;
2445
+ const totalCount = Object.keys(run.results || {}).length;
2446
+ onProgress?.({
2447
+ type: "progress",
2448
+ currentTestCaseIndex: completedCount - 1,
2449
+ totalTestCases: totalCount,
2450
+ currentTestCase: { id: "polling", name: "Polling for status..." }
2451
+ });
2452
+ });
2453
+ if (polledRun) {
2454
+ return polledRun;
2455
+ }
2456
+ }
2457
+ throw streamError;
2458
+ } finally {
2459
+ try {
2460
+ await reader.cancel();
2461
+ } catch {
2462
+ }
2463
+ }
2464
+ if (!finalRun) {
2465
+ if (runId) {
2466
+ console.warn("[ApiClient] SSE stream ended without completion event, polling for status...");
2467
+ const polledRun = await this.pollRunStatus(benchmarkId, runId);
2468
+ if (polledRun) {
2469
+ return polledRun;
2470
+ }
2471
+ }
2472
+ throw new Error("No final run received from server");
2473
+ }
2474
+ return finalRun;
2475
+ }
2476
+ /**
2477
+ * Cancel an in-progress benchmark run
2478
+ */
2479
+ async cancelRun(benchmarkId, runId) {
2480
+ const res = await fetch(
2481
+ `${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmarkId)}/cancel`,
2482
+ {
2483
+ method: "POST",
2484
+ headers: { "Content-Type": "application/json" },
2485
+ body: JSON.stringify({ runId })
2486
+ }
2487
+ );
2488
+ if (!res.ok) {
2489
+ const errorBody = await res.text();
2490
+ throw new Error(`Failed to cancel run: ${errorBody}`);
2491
+ }
2492
+ }
2493
+ /**
2494
+ * Get a specific run from a benchmark by ID.
2495
+ *
2496
+ * Fetches the benchmark and extracts the run with the matching ID.
2497
+ *
2498
+ * @param benchmarkId - The benchmark ID
2499
+ * @param runId - The run ID within the benchmark
2500
+ * @returns The run if found, null otherwise
2501
+ */
2502
+ async getRun(benchmarkId, runId) {
2503
+ const benchmark = await this.getBenchmark(benchmarkId);
2504
+ if (!benchmark) {
2505
+ return null;
2506
+ }
2507
+ return benchmark.runs?.find((r) => r.id === runId) || null;
2508
+ }
2509
+ /**
2510
+ * Poll a run until it reaches a terminal state (completed, failed, cancelled).
2511
+ *
2512
+ * Used as a fallback when SSE stream connection is lost but server continues
2513
+ * processing in the background.
2514
+ *
2515
+ * @param benchmarkId - The benchmark ID
2516
+ * @param runId - The run ID to poll
2517
+ * @param onProgress - Optional callback for progress updates during polling
2518
+ * @param timeoutMs - Maximum time to wait (default: 10 minutes)
2519
+ * @returns The final run state, or null if not found
2520
+ */
2521
+ async pollRunStatus(benchmarkId, runId, onProgress, timeoutMs = 6e5) {
2522
+ const startTime = Date.now();
2523
+ const pollInterval = 5e3;
2524
+ while (Date.now() - startTime < timeoutMs) {
2525
+ const run = await this.getRun(benchmarkId, runId);
2526
+ if (!run) return null;
2527
+ onProgress?.(run);
2528
+ if (run.status && ["completed", "failed", "cancelled"].includes(run.status)) {
2529
+ return run;
2530
+ }
2531
+ await new Promise((resolve5) => setTimeout(resolve5, pollInterval));
2532
+ }
2533
+ return this.getRun(benchmarkId, runId);
2534
+ }
2535
+ /**
2536
+ * Get a single report (TestCaseRun) by ID.
2537
+ *
2538
+ * This is the preferred method for fetching reports - use report IDs
2539
+ * from run.results[testCaseId].reportId.
2540
+ *
2541
+ * @param reportId - The report ID to fetch
2542
+ * @returns The report if found, null otherwise
2543
+ */
2544
+ async getReportById(reportId) {
2545
+ const res = await fetch(
2546
+ `${this.baseUrl}/api/storage/runs/${encodeURIComponent(reportId)}`
2547
+ );
2548
+ if (res.status === 404) {
2549
+ return null;
2550
+ }
2551
+ if (!res.ok) {
2552
+ throw new Error(`Failed to get report: ${res.status} ${res.statusText}`);
2553
+ }
2554
+ return res.json();
2555
+ }
2556
+ /**
2557
+ * List all test cases (basic)
2558
+ */
2559
+ async listTestCases() {
2560
+ const res = await fetch(`${this.baseUrl}/api/storage/test-cases`);
2561
+ if (!res.ok) {
2562
+ throw new Error(`Failed to list test cases: ${res.status} ${res.statusText}`);
2563
+ }
2564
+ const data = await res.json();
2565
+ return data.testCases || [];
2566
+ }
2567
+ /**
2568
+ * List all test cases with full data and metadata
2569
+ */
2570
+ async listTestCasesWithMeta() {
2571
+ const res = await fetch(`${this.baseUrl}/api/storage/test-cases`);
2572
+ if (!res.ok) {
2573
+ throw new Error(`Failed to list test cases: ${res.status} ${res.statusText}`);
2574
+ }
2575
+ const data = await res.json();
2576
+ return {
2577
+ data: data.testCases || [],
2578
+ total: data.total || 0,
2579
+ meta: data.meta || {
2580
+ storageConfigured: false,
2581
+ storageReachable: false,
2582
+ realDataCount: 0,
2583
+ sampleDataCount: data.testCases?.length || 0
2584
+ }
2585
+ };
2586
+ }
2587
+ /**
2588
+ * Bulk create test cases
2589
+ */
2590
+ async bulkCreateTestCases(testCases) {
2591
+ const res = await fetch(`${this.baseUrl}/api/storage/test-cases/bulk`, {
2592
+ method: "POST",
2593
+ headers: { "Content-Type": "application/json" },
2594
+ body: JSON.stringify({ testCases })
2595
+ });
2596
+ if (!res.ok) {
2597
+ const errorBody = await res.text();
2598
+ let errorMessage;
2599
+ try {
2600
+ const parsed = JSON.parse(errorBody);
2601
+ errorMessage = parsed.error || errorBody;
2602
+ } catch {
2603
+ errorMessage = errorBody;
2604
+ }
2605
+ throw new Error(`Failed to bulk create test cases: ${errorMessage}`);
2606
+ }
2607
+ return res.json();
2608
+ }
2609
+ /**
2610
+ * List all benchmarks with metadata
2611
+ */
2612
+ async listBenchmarksWithMeta() {
2613
+ const res = await fetch(`${this.baseUrl}/api/storage/benchmarks`);
2614
+ if (!res.ok) {
2615
+ throw new Error(`Failed to list benchmarks: ${res.status} ${res.statusText}`);
2616
+ }
2617
+ const data = await res.json();
2618
+ return {
2619
+ data: data.benchmarks || [],
2620
+ total: data.total || 0,
2621
+ meta: data.meta || {
2622
+ storageConfigured: false,
2623
+ storageReachable: false,
2624
+ realDataCount: 0,
2625
+ sampleDataCount: data.benchmarks?.length || 0
2626
+ }
2627
+ };
2628
+ }
2629
+ /**
2630
+ * List all configured agents
2631
+ */
2632
+ async listAgents() {
2633
+ const res = await fetch(`${this.baseUrl}/api/agents`);
2634
+ if (!res.ok) {
2635
+ throw new Error(`Failed to list agents: ${res.status} ${res.statusText}`);
2636
+ }
2637
+ const data = await res.json();
2638
+ return data.agents || [];
2639
+ }
2640
+ /**
2641
+ * List all configured models
2642
+ */
2643
+ async listModels() {
2644
+ const res = await fetch(`${this.baseUrl}/api/models`);
2645
+ if (!res.ok) {
2646
+ throw new Error(`Failed to list models: ${res.status} ${res.statusText}`);
2647
+ }
2648
+ const data = await res.json();
2649
+ return data.models || [];
2650
+ }
2651
+ /**
2652
+ * Get a single test case by ID
2653
+ */
2654
+ async getTestCase(id) {
2655
+ const res = await fetch(`${this.baseUrl}/api/storage/test-cases/${encodeURIComponent(id)}`);
2656
+ if (res.status === 404) {
2657
+ return null;
2658
+ }
2659
+ if (!res.ok) {
2660
+ throw new Error(`Failed to get test case: ${res.status} ${res.statusText}`);
2661
+ }
2662
+ return res.json();
2663
+ }
2664
+ /**
2665
+ * Find test case by ID or name
2666
+ */
2667
+ async findTestCase(identifier) {
2668
+ const byId = await this.getTestCase(identifier);
2669
+ if (byId) {
2670
+ return byId;
2671
+ }
2672
+ const response = await this.listTestCasesWithMeta();
2673
+ return response.data.find(
2674
+ (tc) => tc.name.toLowerCase() === identifier.toLowerCase()
2675
+ ) || null;
2676
+ }
2677
+ /**
2678
+ * Evaluation progress event types
2679
+ */
2680
+ /**
2681
+ * Run a single test case evaluation via server API (SSE stream)
2682
+ */
2683
+ async runEvaluation(testCaseId, agentKey, modelId, onProgress) {
2684
+ const res = await fetch(`${this.baseUrl}/api/evaluate`, {
2685
+ method: "POST",
2686
+ headers: { "Content-Type": "application/json" },
2687
+ body: JSON.stringify({ testCaseId, agentKey, modelId })
2688
+ });
2689
+ if (!res.ok) {
2690
+ const errorBody = await res.text();
2691
+ let errorMessage;
2692
+ try {
2693
+ const parsed = JSON.parse(errorBody);
2694
+ errorMessage = parsed.error || errorBody;
2695
+ } catch {
2696
+ errorMessage = errorBody;
2697
+ }
2698
+ throw new Error(`Failed to run evaluation: ${errorMessage}`);
2699
+ }
2700
+ if (!res.body) {
2701
+ throw new Error("Response body is missing");
2702
+ }
2703
+ const reader = res.body.getReader();
2704
+ const decoder = new TextDecoder();
2705
+ let buffer = "";
2706
+ let result = null;
2707
+ try {
2708
+ while (true) {
2709
+ const { done, value } = await reader.read();
2710
+ if (done) break;
2711
+ buffer += decoder.decode(value, { stream: true });
2712
+ const lines = buffer.split("\n\n");
2713
+ buffer = lines.pop() || "";
2714
+ for (const line of lines) {
2715
+ if (line.startsWith("data: ")) {
2716
+ try {
2717
+ const event = JSON.parse(line.slice(6));
2718
+ onProgress?.(event);
2719
+ if (event.type === "completed") {
2720
+ result = event.report;
2721
+ } else if (event.type === "error") {
2722
+ throw new Error(event.error);
2723
+ }
2724
+ } catch (e) {
2725
+ if (e instanceof SyntaxError) continue;
2726
+ throw e;
2727
+ }
2728
+ }
2729
+ }
2730
+ }
2731
+ } finally {
2732
+ try {
2733
+ await reader.cancel();
2734
+ } catch {
2735
+ }
2736
+ }
2737
+ if (!result) {
2738
+ throw new Error("No result received from evaluation");
2739
+ }
2740
+ return result;
2741
+ }
2742
+ /**
2743
+ * Export benchmark test cases as import-compatible JSON
2744
+ */
2745
+ async exportBenchmark(benchmarkId) {
2746
+ const res = await fetch(
2747
+ `${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmarkId)}/export`
2748
+ );
2749
+ if (res.status === 404) {
2750
+ throw new Error(`Benchmark not found: ${benchmarkId}`);
2751
+ }
2752
+ if (!res.ok) {
2753
+ throw new Error(`Failed to export benchmark: ${res.status} ${res.statusText}`);
2754
+ }
2755
+ return res.json();
2756
+ }
2757
+ /**
2758
+ * Create a new benchmark
2759
+ */
2760
+ async createBenchmark(input) {
2761
+ const res = await fetch(`${this.baseUrl}/api/storage/benchmarks`, {
2762
+ method: "POST",
2763
+ headers: { "Content-Type": "application/json" },
2764
+ body: JSON.stringify(input)
2765
+ });
2766
+ if (!res.ok) {
2767
+ const errorBody = await res.text();
2768
+ throw new Error(`Failed to create benchmark: ${errorBody}`);
2769
+ }
2770
+ return res.json();
2771
+ }
2772
+ };
2773
+
2774
+ // cli/commands/list.ts
2775
+ function formatJson(data) {
2776
+ return JSON.stringify(data, null, 2);
2777
+ }
2778
+ function displayStorageWarnings(meta) {
2779
+ if (!meta.storageConfigured) {
2780
+ console.log(chalk.yellow("\n \u26A0 Storage not configured"));
2781
+ console.log(chalk.gray(" Showing sample data only. Set OPENSEARCH_STORAGE_* env vars for persistent storage.\n"));
2782
+ } else if (!meta.storageReachable) {
2783
+ console.log(chalk.yellow("\n \u26A0 Storage unreachable"));
2784
+ meta.warnings?.forEach((w) => console.log(chalk.gray(` - ${w}`)));
2785
+ console.log();
2786
+ }
2787
+ }
2788
+ async function listAgents(format, config) {
2789
+ const serverResult = await ensureServer(config.server);
2790
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
2791
+ try {
2792
+ const client = new ApiClient(serverResult.baseUrl);
2793
+ const agents = await client.listAgents();
2794
+ if (format === "json") {
2795
+ console.log(formatJson(agents));
2796
+ return;
2797
+ }
2798
+ const table = new Table({
2799
+ head: [
2800
+ chalk.cyan("Key"),
2801
+ chalk.cyan("Name"),
2802
+ chalk.cyan("Connector"),
2803
+ chalk.cyan("Models"),
2804
+ chalk.cyan("Endpoint")
2805
+ ],
2806
+ colWidths: [15, 20, 15, 25, 40],
2807
+ wordWrap: true
2808
+ });
2809
+ for (const agent of agents) {
2810
+ table.push([
2811
+ agent.key,
2812
+ agent.name,
2813
+ agent.connectorType || "agui-streaming",
2814
+ agent.models.slice(0, 3).join(", ") + (agent.models.length > 3 ? "..." : ""),
2815
+ agent.endpoint.substring(0, 37) + (agent.endpoint.length > 37 ? "..." : "")
2816
+ ]);
2817
+ }
2818
+ console.log(chalk.bold("\nAvailable Agents:\n"));
2819
+ console.log(table.toString());
2820
+ console.log(chalk.gray(`
2821
+ Total: ${agents.length} agents
2822
+ `));
2823
+ } catch (error) {
2824
+ console.error(chalk.red(`
2825
+ Error: ${error.message}`));
2826
+ console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
2827
+ process.exit(1);
2828
+ } finally {
2829
+ cleanup();
2830
+ }
2831
+ }
2832
+ async function listTestCases(format, config) {
2833
+ const serverResult = await ensureServer(config.server);
2834
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
2835
+ try {
2836
+ const client = new ApiClient(serverResult.baseUrl);
2837
+ const response = await client.listTestCasesWithMeta();
2838
+ displayStorageWarnings(response.meta);
2839
+ if (format === "json") {
2840
+ console.log(formatJson(response));
2841
+ return;
2842
+ }
2843
+ const table = new Table({
2844
+ head: [
2845
+ chalk.cyan("ID"),
2846
+ chalk.cyan("Name"),
2847
+ chalk.cyan("Labels"),
2848
+ chalk.cyan("Version"),
2849
+ chalk.cyan("Source")
2850
+ ],
2851
+ colWidths: [25, 28, 28, 10, 10],
2852
+ wordWrap: true
2853
+ });
2854
+ for (const tc of response.data) {
2855
+ const isDemo = tc.id.startsWith("demo-");
2856
+ table.push([
2857
+ tc.id,
2858
+ tc.name,
2859
+ tc.labels?.slice(0, 3).join(", ") || "",
2860
+ `v${tc.currentVersion || 1}`,
2861
+ isDemo ? chalk.gray("Sample") : chalk.green("Stored")
2862
+ ]);
2863
+ }
2864
+ console.log(chalk.bold("\nAvailable Test Cases:\n"));
2865
+ console.log(table.toString());
2866
+ const { meta } = response;
2867
+ if (meta.realDataCount > 0 || meta.sampleDataCount > 0) {
2868
+ console.log(chalk.gray(`
2869
+ Total: ${response.total} test cases (${meta.realDataCount} stored, ${meta.sampleDataCount} sample)
2870
+ `));
2871
+ } else {
2872
+ console.log(chalk.gray(`
2873
+ Total: ${response.total} test cases
2874
+ `));
2875
+ }
2876
+ } catch (error) {
2877
+ console.error(chalk.red(`
2878
+ Error: ${error.message}`));
2879
+ console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
2880
+ process.exit(1);
2881
+ } finally {
2882
+ cleanup();
2883
+ }
2884
+ }
2885
+ async function listBenchmarks(format, config) {
2886
+ const serverResult = await ensureServer(config.server);
2887
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
2888
+ try {
2889
+ const client = new ApiClient(serverResult.baseUrl);
2890
+ const response = await client.listBenchmarksWithMeta();
2891
+ displayStorageWarnings(response.meta);
2892
+ if (format === "json") {
2893
+ console.log(formatJson(response));
2894
+ return;
2895
+ }
2896
+ const table = new Table({
2897
+ head: [
2898
+ chalk.cyan("ID"),
2899
+ chalk.cyan("Name"),
2900
+ chalk.cyan("Test Cases"),
2901
+ chalk.cyan("Created"),
2902
+ chalk.cyan("Source")
2903
+ ],
2904
+ colWidths: [28, 28, 12, 22, 10],
2905
+ wordWrap: true
2906
+ });
2907
+ for (const b of response.data) {
2908
+ const isDemo = b.id.startsWith("demo-");
2909
+ table.push([
2910
+ b.id,
2911
+ b.name,
2912
+ b.testCaseIds.length.toString(),
2913
+ new Date(b.createdAt).toLocaleDateString(),
2914
+ isDemo ? chalk.gray("Sample") : chalk.green("Stored")
2915
+ ]);
2916
+ }
2917
+ console.log(chalk.bold("\nAvailable Benchmarks:\n"));
2918
+ console.log(table.toString());
2919
+ const { meta } = response;
2920
+ if (meta.realDataCount > 0 || meta.sampleDataCount > 0) {
2921
+ console.log(chalk.gray(`
2922
+ Total: ${response.total} benchmarks (${meta.realDataCount} stored, ${meta.sampleDataCount} sample)
2923
+ `));
2924
+ } else {
2925
+ console.log(chalk.gray(`
2926
+ Total: ${response.total} benchmarks
2927
+ `));
2928
+ }
2929
+ } catch (error) {
2930
+ console.error(chalk.red(`
2931
+ Error: ${error.message}`));
2932
+ console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
2933
+ process.exit(1);
2934
+ } finally {
2935
+ cleanup();
2936
+ }
2937
+ }
2938
+ function listConnectors(format) {
2939
+ const types = connectorRegistry.getRegisteredTypes();
2940
+ const connectors = types.map((type) => {
2941
+ const connector = connectorRegistry.get(type);
2942
+ return {
2943
+ type,
2944
+ name: connector?.name || "Unknown",
2945
+ streaming: connector?.supportsStreaming || false
2946
+ };
2947
+ });
2948
+ if (format === "json") {
2949
+ console.log(formatJson(connectors));
2950
+ return;
2951
+ }
2952
+ const table = new Table({
2953
+ head: [
2954
+ chalk.cyan("Type"),
2955
+ chalk.cyan("Name"),
2956
+ chalk.cyan("Streaming")
2957
+ ],
2958
+ colWidths: [20, 25, 12]
2959
+ });
2960
+ for (const c of connectors) {
2961
+ table.push([
2962
+ c.type,
2963
+ c.name,
2964
+ c.streaming ? chalk.green("Yes") : chalk.gray("No")
2965
+ ]);
2966
+ }
2967
+ console.log(chalk.bold("\nRegistered Connectors:\n"));
2968
+ console.log(table.toString());
2969
+ console.log(chalk.gray(`
2970
+ Total: ${connectors.length} connectors
2971
+ `));
2972
+ }
2973
+ async function listModels(format, config) {
2974
+ const serverResult = await ensureServer(config.server);
2975
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
2976
+ try {
2977
+ const client = new ApiClient(serverResult.baseUrl);
2978
+ const models = await client.listModels();
2979
+ if (format === "json") {
2980
+ console.log(formatJson(models));
2981
+ return;
2982
+ }
2983
+ const table = new Table({
2984
+ head: [
2985
+ chalk.cyan("Key"),
2986
+ chalk.cyan("Display Name"),
2987
+ chalk.cyan("Provider"),
2988
+ chalk.cyan("Context")
2989
+ ],
2990
+ colWidths: [25, 30, 12, 12],
2991
+ wordWrap: true
2992
+ });
2993
+ for (const m of models) {
2994
+ table.push([
2995
+ m.key,
2996
+ m.display_name || m.key,
2997
+ m.provider || "bedrock",
2998
+ m.context_window ? `${Math.round(m.context_window / 1e3)}k` : "-"
2999
+ ]);
3000
+ }
3001
+ console.log(chalk.bold("\nAvailable Models:\n"));
3002
+ console.log(table.toString());
3003
+ console.log(chalk.gray(`
3004
+ Total: ${models.length} models
3005
+ `));
3006
+ } catch (error) {
3007
+ console.error(chalk.red(`
3008
+ Error: ${error.message}`));
3009
+ console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
3010
+ process.exit(1);
3011
+ } finally {
3012
+ cleanup();
3013
+ }
3014
+ }
3015
+ function createListCommand() {
3016
+ const command = new Command("list").description("List available resources").argument("<resource>", "Resource type: agents, test-cases, benchmarks, connectors, models").option("-o, --output <format>", "Output format: table, json", "table").action(async (resource, options) => {
3017
+ const format = options.output;
3018
+ const config = await loadConfig();
3019
+ for (const connector of config.connectors) {
3020
+ connectorRegistry.register(connector);
3021
+ }
3022
+ switch (resource.toLowerCase()) {
3023
+ case "agents":
3024
+ await listAgents(format, config);
3025
+ break;
3026
+ case "test-cases":
3027
+ case "testcases":
3028
+ case "tc":
3029
+ await listTestCases(format, config);
3030
+ break;
3031
+ case "benchmarks":
3032
+ case "bench":
3033
+ await listBenchmarks(format, config);
3034
+ break;
3035
+ case "connectors":
3036
+ listConnectors(format);
3037
+ break;
3038
+ case "models":
3039
+ await listModels(format, config);
3040
+ break;
3041
+ default:
3042
+ console.error(chalk.red(`
3043
+ Unknown resource type: ${resource}`));
3044
+ console.log(chalk.gray(" Available: agents, test-cases, benchmarks, connectors, models\n"));
3045
+ process.exit(1);
3046
+ }
3047
+ });
3048
+ return command;
3049
+ }
3050
+
3051
+ // cli/commands/run.ts
3052
+ import { Command as Command2 } from "commander";
3053
+ import chalk2 from "chalk";
3054
+ import ora from "ora";
3055
+ import Table2 from "cli-table3";
3056
+ function findAgent(identifier, config) {
3057
+ return config.agents.find(
3058
+ (a) => a.key === identifier || a.name.toLowerCase() === identifier.toLowerCase()
3059
+ );
3060
+ }
3061
+ function getDefaultModel(agent) {
3062
+ return agent.models[0] || "claude-sonnet";
3063
+ }
3064
+ async function commandExists(command) {
3065
+ const { execSync: execSync2 } = await import("child_process");
3066
+ const checkCommand = process.platform === "win32" ? `where ${command}` : `which ${command}`;
3067
+ try {
3068
+ execSync2(checkCommand, { stdio: "ignore" });
3069
+ return true;
3070
+ } catch {
3071
+ return false;
3072
+ }
3073
+ }
3074
+ async function validateAgentRequirements(agent) {
3075
+ if (agent.connectorType === "claude-code") {
3076
+ if (!await commandExists("claude")) {
3077
+ return `Claude CLI not found in PATH. Install Claude Code CLI: https://claude.ai/code`;
3078
+ }
3079
+ }
3080
+ if (agent.connectorType === "subprocess" && agent.endpoint) {
3081
+ const command = agent.endpoint.split(" ")[0];
3082
+ if (!await commandExists(command)) {
3083
+ return `Command '${command}' not found in PATH`;
3084
+ }
3085
+ }
3086
+ return void 0;
3087
+ }
3088
+ async function runForAgent(client, testCaseId, agent, modelId, verbose) {
3089
+ const spinner = ora(`Running ${agent.name}...`).start();
3090
+ try {
3091
+ const report = await client.runEvaluation(
3092
+ testCaseId,
3093
+ agent.key,
3094
+ modelId,
3095
+ (event) => {
3096
+ if (event.type === "step" && verbose) {
3097
+ spinner.text = `${agent.name}: Step ${event.stepIndex + 1} (${event.step.type})`;
3098
+ } else if (event.type === "started") {
3099
+ spinner.text = `${agent.name}: Started evaluation...`;
3100
+ }
3101
+ }
3102
+ );
3103
+ if (report.status === "completed" && report.passFailStatus === "passed") {
3104
+ spinner.succeed(`${agent.name}: ${chalk2.green("PASSED")}`);
3105
+ } else if (report.status === "completed") {
3106
+ spinner.succeed(`${agent.name}: ${chalk2.red("FAILED")}`);
3107
+ } else {
3108
+ spinner.fail(`${agent.name}: ${chalk2.yellow(report.status)}`);
3109
+ }
3110
+ return report;
3111
+ } catch (error) {
3112
+ spinner.fail(`${agent.name}: ${chalk2.red("ERROR")}`);
3113
+ throw error;
3114
+ }
3115
+ }
3116
+ function displayTableResults(results) {
3117
+ const table = new Table2({
3118
+ head: [
3119
+ chalk2.cyan("Agent"),
3120
+ chalk2.cyan("Status"),
3121
+ chalk2.cyan("Accuracy"),
3122
+ chalk2.cyan("Steps"),
3123
+ chalk2.cyan("Report ID")
3124
+ ],
3125
+ colWidths: [20, 12, 12, 10, 30]
3126
+ });
3127
+ for (const r of results) {
3128
+ if (!r.report) {
3129
+ table.push([
3130
+ r.agent.name,
3131
+ chalk2.red("ERROR"),
3132
+ "-",
3133
+ "-",
3134
+ "-"
3135
+ ]);
3136
+ continue;
3137
+ }
3138
+ const statusStr = r.report.passFailStatus === "passed" ? chalk2.green("PASSED") : r.report.passFailStatus === "failed" ? chalk2.red("FAILED") : chalk2.yellow(r.report.status);
3139
+ table.push([
3140
+ r.agent.name,
3141
+ statusStr,
3142
+ r.report.metrics?.accuracy ? `${Math.round(r.report.metrics.accuracy)}%` : "-",
3143
+ r.report.trajectorySteps.toString(),
3144
+ r.report.id?.substring(0, 27) + "..." || "-"
3145
+ ]);
3146
+ }
3147
+ console.log("\n");
3148
+ console.log(table.toString());
3149
+ }
3150
+ function createRunCommand() {
3151
+ const command = new Command2("run").description("Run a test case against agents").requiredOption("-t, --test-case <id>", "Test case ID or name").option("-a, --agent <key>", "Agent key (can be specified multiple times)", (val, arr) => [...arr, val], []).option("-m, --model <id>", "Model ID (uses agent default if not specified)").option("-o, --output <format>", "Output format: table, json", "table").option("-v, --verbose", "Show detailed trajectory output").action(async (options) => {
3152
+ console.log(chalk2.bold("\nAgent Health - Test Case Runner\n"));
3153
+ const config = await loadConfig();
3154
+ for (const connector of config.connectors) {
3155
+ connectorRegistry.register(connector);
3156
+ }
3157
+ const serverResult = await ensureServer(config.server);
3158
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
3159
+ try {
3160
+ const client = new ApiClient(serverResult.baseUrl);
3161
+ await client.checkHealth();
3162
+ const testCase = await client.findTestCase(options.testCase);
3163
+ if (!testCase) {
3164
+ console.error(chalk2.red(` Error: Test case not found: ${options.testCase}`));
3165
+ console.log(chalk2.gray(" Use `agent-health list test-cases` to see available test cases\n"));
3166
+ process.exit(1);
3167
+ }
3168
+ console.log(chalk2.gray(` Test Case: ${testCase.name} (${testCase.id})`));
3169
+ console.log(chalk2.gray(` Server: ${serverResult.baseUrl}`));
3170
+ let agents = [];
3171
+ if (options.agent.length === 0) {
3172
+ agents = [config.agents[0]];
3173
+ console.log(chalk2.gray(` Agent: ${agents[0].name} (default)`));
3174
+ } else {
3175
+ for (const agentId of options.agent) {
3176
+ const agent = findAgent(agentId, config);
3177
+ if (!agent) {
3178
+ console.error(chalk2.red(` Error: Agent not found: ${agentId}`));
3179
+ console.log(chalk2.gray(" Use `agent-health list agents` to see available agents\n"));
3180
+ process.exit(1);
3181
+ }
3182
+ agents.push(agent);
3183
+ }
3184
+ console.log(chalk2.gray(` Agents: ${agents.map((a) => a.name).join(", ")}`));
3185
+ }
3186
+ console.log("");
3187
+ const results = [];
3188
+ for (const agent of agents) {
3189
+ const modelId = options.model || getDefaultModel(agent);
3190
+ const validationError = await validateAgentRequirements(agent);
3191
+ if (validationError) {
3192
+ console.error(chalk2.red(` Error: ${validationError}`));
3193
+ results.push({ agent, report: null });
3194
+ continue;
3195
+ }
3196
+ try {
3197
+ const report = await runForAgent(client, testCase.id, agent, modelId, options.verbose || false);
3198
+ results.push({ agent, report });
3199
+ } catch (error) {
3200
+ console.error(chalk2.red(` Error running ${agent.name}: ${error instanceof Error ? error.message : error}`));
3201
+ results.push({ agent, report: null });
3202
+ }
3203
+ }
3204
+ if (options.output === "json") {
3205
+ console.log(JSON.stringify(results.map((r) => ({
3206
+ agent: { key: r.agent.key, name: r.agent.name },
3207
+ report: r.report
3208
+ })), null, 2));
3209
+ } else {
3210
+ displayTableResults(results);
3211
+ }
3212
+ } catch (error) {
3213
+ console.error(chalk2.red(`
3214
+ Error: ${error.message}`));
3215
+ console.log(chalk2.gray(" Is the server running? Start with: npm run dev:server\n"));
3216
+ process.exit(1);
3217
+ } finally {
3218
+ cleanup();
3219
+ }
3220
+ });
3221
+ return command;
3222
+ }
3223
+
3224
+ // cli/commands/benchmark.ts
3225
+ import { Command as Command3 } from "commander";
3226
+ import chalk3 from "chalk";
3227
+ import ora2 from "ora";
3228
+ import Table3 from "cli-table3";
3229
+ import { readFileSync as readFileSync2, writeFileSync } from "fs";
3230
+
3231
+ // lib/testCaseValidation.ts
3232
+ import { z } from "zod";
3233
+ var contextItemSchema = z.object({
3234
+ description: z.string(),
3235
+ value: z.string()
3236
+ }).required();
3237
+ var difficultySchema = z.enum(["Easy", "Medium", "Hard"]);
3238
+ var testCaseSchema = z.object({
3239
+ name: z.string().min(1, "Name is required"),
3240
+ description: z.string().optional().default(""),
3241
+ category: z.string().min(1, "Category is required"),
3242
+ subcategory: z.string().optional(),
3243
+ difficulty: difficultySchema,
3244
+ initialPrompt: z.string().min(1, "Initial prompt is required"),
3245
+ context: z.array(contextItemSchema).optional().default([]),
3246
+ expectedOutcomes: z.array(z.string()).refine(
3247
+ (outcomes) => outcomes.some((o) => o.trim().length > 0),
3248
+ "At least one non-empty expected outcome is required"
3249
+ )
3250
+ });
3251
+ var testCasesArraySchema = z.array(testCaseSchema).min(1, "Array cannot be empty");
3252
+ function zodErrorToValidationErrors(error) {
3253
+ return error.errors.map((e) => ({
3254
+ path: e.path.join("."),
3255
+ message: e.message,
3256
+ type: "error"
3257
+ }));
3258
+ }
3259
+ function validateTestCaseJson(json) {
3260
+ if (Array.isArray(json)) {
3261
+ return {
3262
+ valid: false,
3263
+ errors: [{ path: "", message: "Received an array. Use Bulk Import mode for multiple test cases", type: "error" }]
3264
+ };
3265
+ }
3266
+ const result = testCaseSchema.safeParse(json);
3267
+ if (!result.success) {
3268
+ return {
3269
+ valid: false,
3270
+ errors: zodErrorToValidationErrors(result.error)
3271
+ };
3272
+ }
3273
+ return { valid: true, errors: [], data: result.data };
3274
+ }
3275
+ function validateTestCasesArrayJson(json) {
3276
+ if (json && typeof json === "object" && !Array.isArray(json)) {
3277
+ const singleResult = validateTestCaseJson(json);
3278
+ if (singleResult.valid && singleResult.data) {
3279
+ return {
3280
+ valid: true,
3281
+ errors: [],
3282
+ data: [singleResult.data]
3283
+ };
3284
+ }
3285
+ return { valid: false, errors: singleResult.errors };
3286
+ }
3287
+ const result = testCasesArraySchema.safeParse(json);
3288
+ if (!result.success) {
3289
+ return {
3290
+ valid: false,
3291
+ errors: zodErrorToValidationErrors(result.error)
3292
+ };
3293
+ }
3294
+ return { valid: true, errors: [], data: result.data };
3295
+ }
3296
+
3297
+ // lib/runStats.ts
3298
+ function calculateRunStats(run, reports) {
3299
+ let passed = 0;
3300
+ let failed = 0;
3301
+ let pending = 0;
3302
+ let total = 0;
3303
+ Object.entries(run.results || {}).forEach(([testCaseId, result]) => {
3304
+ total++;
3305
+ if (result.status === "pending" || result.status === "running") {
3306
+ pending++;
3307
+ return;
3308
+ }
3309
+ if (result.status === "failed" || result.status === "cancelled") {
3310
+ failed++;
3311
+ return;
3312
+ }
3313
+ if (result.status === "completed" && result.reportId) {
3314
+ const report = reports[result.reportId];
3315
+ if (!report) {
3316
+ pending++;
3317
+ return;
3318
+ }
3319
+ if (report.metricsStatus === "pending" || report.metricsStatus === "calculating") {
3320
+ pending++;
3321
+ return;
3322
+ }
3323
+ if (report.passFailStatus === "passed") {
3324
+ passed++;
3325
+ } else {
3326
+ failed++;
3327
+ }
3328
+ } else {
3329
+ pending++;
3330
+ }
3331
+ });
3332
+ const passRate = total > 0 ? Math.round(passed / total * 100) : 0;
3333
+ return {
3334
+ passed,
3335
+ failed,
3336
+ pending,
3337
+ total,
3338
+ passRate
3339
+ };
3340
+ }
3341
+ function getReportIdsFromRun(run) {
3342
+ const reportIds = /* @__PURE__ */ new Set();
3343
+ Object.values(run.results || {}).forEach((result) => {
3344
+ if (result.reportId) {
3345
+ reportIds.add(result.reportId);
3346
+ }
3347
+ });
3348
+ return Array.from(reportIds);
3349
+ }
3350
+
3351
+ // cli/commands/benchmark.ts
3352
+ function findAgent2(identifier, config) {
3353
+ return config.agents.find(
3354
+ (a) => a.key === identifier || a.name.toLowerCase() === identifier.toLowerCase()
3355
+ );
3356
+ }
3357
+ function getDefaultModel2(agent) {
3358
+ return agent.models[0] || "claude-sonnet";
3359
+ }
3360
+ function isFilePath(value) {
3361
+ return value.toLowerCase().endsWith(".json");
3362
+ }
3363
+ function loadAndValidateTestCasesFile(filePath) {
3364
+ let raw;
3365
+ try {
3366
+ raw = readFileSync2(filePath, "utf-8");
3367
+ } catch (err) {
3368
+ throw new Error(`Cannot read file: ${filePath} (${err instanceof Error ? err.message : err})`);
3369
+ }
3370
+ let parsed;
3371
+ try {
3372
+ parsed = JSON.parse(raw);
3373
+ } catch {
3374
+ throw new Error(`Invalid JSON in file: ${filePath}`);
3375
+ }
3376
+ const result = validateTestCasesArrayJson(parsed);
3377
+ if (!result.valid || !result.data) {
3378
+ const msgs = result.errors.map((e) => e.path ? `${e.path}: ${e.message}` : e.message).join("\n ");
3379
+ throw new Error(`Validation failed for ${filePath}:
3380
+ ${msgs}`);
3381
+ }
3382
+ return result.data;
3383
+ }
3384
+ async function fetchReportsForRun(api, run) {
3385
+ const reportIds = getReportIdsFromRun(run);
3386
+ const reportsMap = {};
3387
+ await Promise.all(
3388
+ reportIds.map(async (reportId) => {
3389
+ reportsMap[reportId] = await api.getReportById(reportId);
3390
+ })
3391
+ );
3392
+ return reportsMap;
3393
+ }
3394
+ async function runBenchmarkForAgent(api, agent, modelId, benchmark, verbose) {
3395
+ const results = {
3396
+ agent,
3397
+ passed: 0,
3398
+ failed: 0
3399
+ };
3400
+ const totalTestCases = benchmark.testCaseIds.length;
3401
+ const spinner = ora2(`Running ${agent.name} (0/${totalTestCases})`).start();
3402
+ let startedRunId;
3403
+ try {
3404
+ const completedRun = await api.executeBenchmark(
3405
+ benchmark.id,
3406
+ {
3407
+ name: `CLI Run - ${agent.name}`,
3408
+ agentKey: agent.key,
3409
+ modelId
3410
+ },
3411
+ (event) => {
3412
+ if (event.type === "started") {
3413
+ startedRunId = event.runId;
3414
+ } else if (event.type === "progress") {
3415
+ const current = event.currentTestCaseIndex + 1;
3416
+ const testCaseName = event.currentTestCase?.name || `Test ${current}`;
3417
+ spinner.text = `${agent.name}: ${testCaseName} (${current}/${totalTestCases})`;
3418
+ if (verbose && event.result) {
3419
+ const status = event.result.status === "completed" ? chalk3.green("\u2713") : chalk3.red("\u2717");
3420
+ spinner.text = `${agent.name}: ${testCaseName} ${status} (${current}/${totalTestCases})`;
3421
+ }
3422
+ }
3423
+ }
3424
+ );
3425
+ results.run = completedRun;
3426
+ const reportsMap = await fetchReportsForRun(api, completedRun);
3427
+ const stats = calculateRunStats(completedRun, reportsMap);
3428
+ results.passed = stats.passed;
3429
+ results.failed = stats.failed;
3430
+ results.reports = Object.values(reportsMap).filter((r) => r !== null);
3431
+ const passRate = stats.passRate;
3432
+ if (passRate >= 80) {
3433
+ spinner.succeed(
3434
+ `${agent.name}: ${chalk3.green(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
3435
+ );
3436
+ } else if (passRate >= 50) {
3437
+ spinner.warn(
3438
+ `${agent.name}: ${chalk3.yellow(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
3439
+ );
3440
+ } else {
3441
+ spinner.fail(
3442
+ `${agent.name}: ${chalk3.red(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
3443
+ );
3444
+ }
3445
+ } catch (error) {
3446
+ const errorMessage = error instanceof Error ? error.message : String(error);
3447
+ if (startedRunId) {
3448
+ results.runId = startedRunId;
3449
+ try {
3450
+ const run = await api.getRun(benchmark.id, startedRunId);
3451
+ if (run) {
3452
+ results.run = run;
3453
+ const reportsMap = await fetchReportsForRun(api, run);
3454
+ const stats = calculateRunStats(run, reportsMap);
3455
+ results.passed = stats.passed;
3456
+ results.failed = stats.failed;
3457
+ results.reports = Object.values(reportsMap).filter((r) => r !== null);
3458
+ if (run.status === "completed" || run.status === "cancelled") {
3459
+ const passRate = stats.passRate;
3460
+ if (passRate >= 80) {
3461
+ spinner.succeed(
3462
+ `${agent.name}: ${chalk3.green(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
3463
+ );
3464
+ } else if (passRate >= 50) {
3465
+ spinner.warn(
3466
+ `${agent.name}: ${chalk3.yellow(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
3467
+ );
3468
+ } else {
3469
+ spinner.fail(
3470
+ `${agent.name}: ${chalk3.red(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
3471
+ );
3472
+ }
3473
+ return results;
3474
+ }
3475
+ }
3476
+ } catch {
3477
+ }
3478
+ }
3479
+ const isStreamError = errorMessage.includes("terminated") || errorMessage.includes("network") || errorMessage.includes("stream") || errorMessage.includes("aborted");
3480
+ if (isStreamError && startedRunId) {
3481
+ spinner.warn(`${agent.name}: ${chalk3.yellow("Stream disconnected")} - server may still be processing`);
3482
+ console.log(chalk3.gray(` Check status: Use the UI to monitor progress`));
3483
+ } else {
3484
+ spinner.fail(`${agent.name}: ${chalk3.red("Failed")} - ${errorMessage}`);
3485
+ }
3486
+ }
3487
+ return results;
3488
+ }
3489
+ function displaySummaryTable(allResults, totalTestCases) {
3490
+ const table = new Table3({
3491
+ head: [
3492
+ chalk3.cyan("Agent"),
3493
+ chalk3.cyan("Passed"),
3494
+ chalk3.cyan("Failed"),
3495
+ chalk3.cyan("Pass Rate"),
3496
+ chalk3.cyan("Run ID")
3497
+ ],
3498
+ colWidths: [25, 10, 10, 12, 35]
3499
+ });
3500
+ for (const results of allResults) {
3501
+ const passRate = totalTestCases > 0 ? results.passed / totalTestCases * 100 : 0;
3502
+ const passRateColor = passRate >= 80 ? chalk3.green : passRate >= 50 ? chalk3.yellow : chalk3.red;
3503
+ table.push([
3504
+ results.agent.name,
3505
+ chalk3.green(results.passed.toString()),
3506
+ chalk3.red(results.failed.toString()),
3507
+ passRateColor(`${passRate.toFixed(0)}%`),
3508
+ results.run?.id || results.runId || chalk3.gray("N/A")
3509
+ ]);
3510
+ }
3511
+ console.log("\n");
3512
+ console.log(chalk3.bold("Benchmark Summary"));
3513
+ console.log(table.toString());
3514
+ }
3515
+ async function exportResults(benchmark, allResults, exportPath, format, serverBaseUrl) {
3516
+ if (format !== "json") {
3517
+ const runIds = allResults.map((r) => r.run?.id || r.runId).filter((id) => !!id);
3518
+ const params = new URLSearchParams({ format });
3519
+ if (runIds.length > 0) {
3520
+ params.set("runIds", runIds.join(","));
3521
+ }
3522
+ const url = `${serverBaseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmark.id)}/report?${params.toString()}`;
3523
+ const response = await fetch(url);
3524
+ if (!response.ok) {
3525
+ const errorBody = await response.json().catch(() => ({ error: "Unknown error" }));
3526
+ console.error(chalk3.red(`
3527
+ Export failed: ${errorBody.error}`));
3528
+ return;
3529
+ }
3530
+ const contentType = response.headers.get("content-type") || "";
3531
+ if (contentType.includes("application/pdf")) {
3532
+ const buffer = Buffer.from(await response.arrayBuffer());
3533
+ writeFileSync(exportPath, buffer);
3534
+ } else {
3535
+ const text = await response.text();
3536
+ writeFileSync(exportPath, text);
3537
+ }
3538
+ } else {
3539
+ const exportData = {
3540
+ benchmark: {
3541
+ id: benchmark.id,
3542
+ name: benchmark.name,
3543
+ testCaseCount: benchmark.testCaseIds.length
3544
+ },
3545
+ runs: allResults.map((r) => ({
3546
+ agent: { key: r.agent.key, name: r.agent.name },
3547
+ runId: r.run?.id || r.runId,
3548
+ status: r.run?.status,
3549
+ passed: r.passed,
3550
+ failed: r.failed,
3551
+ passRate: benchmark.testCaseIds.length > 0 ? r.passed / benchmark.testCaseIds.length * 100 : 0,
3552
+ results: r.run?.results,
3553
+ reports: r.reports
3554
+ })),
3555
+ exportedAt: (/* @__PURE__ */ new Date()).toISOString()
3556
+ };
3557
+ writeFileSync(exportPath, JSON.stringify(exportData, null, 2));
3558
+ }
3559
+ console.log(chalk3.green(`
3560
+ Results exported to: ${exportPath}`));
3561
+ }
3562
+ function createBenchmarkCommand() {
3563
+ const command = new Command3("benchmark").description("Run a benchmark against one or more agents").option("-n, --name <name>", "Benchmark name or ID (optional in quick mode)").option("-f, --file <path>", "JSON file of test cases to import and benchmark").option(
3564
+ "-a, --agent <key>",
3565
+ "Agent key (can be specified multiple times)",
3566
+ (val, arr) => [...arr, val],
3567
+ []
3568
+ ).option("-m, --model <id>", "Model ID (uses agent default if not specified)").option("-o, --output <format>", "Output format: table, json", "table").option("--export <path>", "Export results to file").option("--format <type>", "Report format for --export: json (default), html, pdf", "json").option("-v, --verbose", "Show detailed output").option("--stop-server", "Stop the server after benchmark completes (default: keep running)").action(async (options) => {
3569
+ console.log(chalk3.bold("\nAgent Health - Benchmark Runner\n"));
3570
+ const config = await loadConfig();
3571
+ const serverConfig = { ...DEFAULT_SERVER_CONFIG, ...config.server };
3572
+ const isCI = !!process.env.CI;
3573
+ const serverWasRunning = await isServerRunning(serverConfig.port);
3574
+ const filePath = options.file || (options.name && isFilePath(options.name) ? options.name : void 0);
3575
+ const fileMode = !!filePath;
3576
+ const quickMode = !options.name && !fileMode && !serverWasRunning;
3577
+ if (!options.name && !fileMode && serverWasRunning) {
3578
+ console.error(chalk3.red(" Error: Benchmark name required when server is already running."));
3579
+ console.log("");
3580
+ console.log(chalk3.cyan(" Options:"));
3581
+ console.log(chalk3.gray(' 1. Specify a benchmark: benchmark -n "Name" -a claude-code'));
3582
+ console.log(chalk3.gray(" 2. Import from file: benchmark -f ./test-cases.json -a mock"));
3583
+ console.log(chalk3.gray(" 3. Stop the server and run in quick mode"));
3584
+ console.log(chalk3.gray(" 4. List available: npx agent-health list benchmarks"));
3585
+ console.log("");
3586
+ process.exit(1);
3587
+ }
3588
+ if (fileMode) {
3589
+ console.log(chalk3.cyan(` Running in file mode (importing test cases from ${filePath})`));
3590
+ } else if (quickMode) {
3591
+ console.log(chalk3.cyan(" Running in quick mode (auto-creating benchmark from test cases)"));
3592
+ }
3593
+ const connectSpinner = ora2("Connecting to server...").start();
3594
+ let serverResult;
3595
+ let cleanup;
3596
+ const shouldStopServer = isCI || quickMode || fileMode || options.stopServer;
3597
+ try {
3598
+ serverResult = await ensureServer(serverConfig);
3599
+ cleanup = createServerCleanup(serverResult, shouldStopServer);
3600
+ if (serverResult.wasStarted) {
3601
+ connectSpinner.succeed(`Started server on port ${serverConfig.port}`);
3602
+ } else {
3603
+ connectSpinner.succeed(`Connected to existing server on port ${serverConfig.port}`);
3604
+ }
3605
+ } catch (error) {
3606
+ connectSpinner.fail(
3607
+ `Failed to connect to server: ${error instanceof Error ? error.message : error}`
3608
+ );
3609
+ process.exit(1);
3610
+ }
3611
+ const api = new ApiClient(serverResult.baseUrl);
3612
+ try {
3613
+ let benchmark = null;
3614
+ if (fileMode) {
3615
+ const importSpinner = ora2(`Loading test cases from ${filePath}...`).start();
3616
+ try {
3617
+ const validatedTestCases = loadAndValidateTestCasesFile(filePath);
3618
+ importSpinner.succeed(`Validated ${validatedTestCases.length} test cases from file`);
3619
+ const uploadSpinner = ora2("Importing test cases to server...").start();
3620
+ const bulkResult = await api.bulkCreateTestCases(validatedTestCases);
3621
+ uploadSpinner.succeed(`Imported ${bulkResult.created} test cases`);
3622
+ const benchmarkName = options.file && options.name ? options.name : `file-${Date.now()}`;
3623
+ const createSpinner = ora2("Creating benchmark...").start();
3624
+ benchmark = await api.createBenchmark({
3625
+ name: benchmarkName,
3626
+ description: `Imported from ${filePath}`,
3627
+ testCaseIds: bulkResult.testCases.map((tc) => tc.id)
3628
+ });
3629
+ createSpinner.succeed(`Created benchmark: ${benchmark.name}`);
3630
+ } catch (error) {
3631
+ importSpinner.fail(`File import failed: ${error instanceof Error ? error.message : error}`);
3632
+ process.exit(1);
3633
+ }
3634
+ } else if (quickMode) {
3635
+ const testCasesSpinner = ora2("Fetching test cases...").start();
3636
+ try {
3637
+ const testCases = await api.listTestCases();
3638
+ if (testCases.length === 0) {
3639
+ testCasesSpinner.fail("No test cases found");
3640
+ console.log(chalk3.gray(" Add test cases via the UI or provide a file with -f option."));
3641
+ process.exit(1);
3642
+ }
3643
+ testCasesSpinner.succeed(`Found ${testCases.length} test cases`);
3644
+ const createSpinner = ora2("Creating quick benchmark...").start();
3645
+ benchmark = await api.createBenchmark({
3646
+ name: `quick-${Date.now()}`,
3647
+ description: "Auto-generated benchmark for quick mode",
3648
+ testCaseIds: testCases.map((tc) => tc.id)
3649
+ });
3650
+ createSpinner.succeed(`Created benchmark: ${benchmark.name}`);
3651
+ } catch (error) {
3652
+ testCasesSpinner.fail(`Failed to create benchmark: ${error instanceof Error ? error.message : error}`);
3653
+ process.exit(1);
3654
+ }
3655
+ } else {
3656
+ benchmark = await api.findBenchmark(options.name);
3657
+ if (!benchmark) {
3658
+ console.error(chalk3.red(` Error: Benchmark not found: "${options.name}"`));
3659
+ console.log("");
3660
+ console.log(chalk3.cyan(" The -n/--name option accepts:"));
3661
+ console.log(chalk3.gray(" \u2022 Benchmark ID (e.g., demo-baseline)"));
3662
+ console.log(chalk3.gray(' \u2022 Benchmark name (case-sensitive, e.g., "Baseline")'));
3663
+ console.log("");
3664
+ console.log(chalk3.cyan(" Or import from file:"));
3665
+ console.log(chalk3.gray(" benchmark -f ./test-cases.json -a mock"));
3666
+ console.log("");
3667
+ console.log(chalk3.cyan(" Available benchmarks:"));
3668
+ console.log(chalk3.gray(" npx agent-health list benchmarks"));
3669
+ console.log("");
3670
+ process.exit(1);
3671
+ }
3672
+ if (benchmark.id.startsWith("demo-")) {
3673
+ console.error(chalk3.red(` Error: Cannot execute sample benchmarks.`));
3674
+ console.log(chalk3.gray(" Sample data is read-only with pre-completed runs."));
3675
+ console.log(chalk3.gray(" Create a real benchmark in the UI to run evaluations."));
3676
+ console.log("");
3677
+ process.exit(1);
3678
+ }
3679
+ }
3680
+ console.log(chalk3.gray(` Benchmark: ${benchmark.name} (${benchmark.id})`));
3681
+ console.log(chalk3.gray(` Test Cases: ${benchmark.testCaseIds.length}`));
3682
+ console.log(chalk3.gray(` Server: ${serverResult.baseUrl}`));
3683
+ let agents = [];
3684
+ if (options.agent.length === 0) {
3685
+ const enabledAgent = config.agents.find((a) => a.enabled !== false);
3686
+ if (!enabledAgent) {
3687
+ console.error(chalk3.red(" Error: No enabled agents found in config."));
3688
+ process.exit(1);
3689
+ }
3690
+ agents = [enabledAgent];
3691
+ console.log(chalk3.gray(` Agent: ${agents[0].name} (default)`));
3692
+ } else {
3693
+ for (const agentId of options.agent) {
3694
+ const agent = findAgent2(agentId, config);
3695
+ if (!agent) {
3696
+ console.error(chalk3.red(` Error: Agent not found: ${agentId}`));
3697
+ console.log(chalk3.gray(" Available agents:"));
3698
+ for (const a of config.agents) {
3699
+ console.log(chalk3.gray(` - ${a.name} (${a.key})`));
3700
+ }
3701
+ console.log("");
3702
+ process.exit(1);
3703
+ }
3704
+ agents.push(agent);
3705
+ }
3706
+ console.log(chalk3.gray(` Agents: ${agents.map((a) => a.name).join(", ")}`));
3707
+ }
3708
+ console.log("");
3709
+ const allResults = [];
3710
+ for (const agent of agents) {
3711
+ const modelId = options.model || getDefaultModel2(agent);
3712
+ const results = await runBenchmarkForAgent(
3713
+ api,
3714
+ agent,
3715
+ modelId,
3716
+ benchmark,
3717
+ options.verbose || false
3718
+ );
3719
+ allResults.push(results);
3720
+ }
3721
+ if (options.output === "json") {
3722
+ const jsonOutput = allResults.map((r) => ({
3723
+ agent: { key: r.agent.key, name: r.agent.name },
3724
+ runId: r.run?.id || r.runId,
3725
+ passed: r.passed,
3726
+ failed: r.failed,
3727
+ passRate: benchmark.testCaseIds.length > 0 ? r.passed / benchmark.testCaseIds.length * 100 : 0,
3728
+ results: r.run?.results
3729
+ }));
3730
+ console.log(JSON.stringify(jsonOutput, null, 2));
3731
+ } else {
3732
+ displaySummaryTable(allResults, benchmark.testCaseIds.length);
3733
+ }
3734
+ if (options.export) {
3735
+ await exportResults(benchmark, allResults, options.export, options.format, serverResult.baseUrl);
3736
+ }
3737
+ console.log("");
3738
+ console.log(chalk3.cyan("View results:"));
3739
+ for (const result of allResults) {
3740
+ const runId = result.run?.id || result.runId;
3741
+ if (runId) {
3742
+ console.log(chalk3.gray(` ${result.agent.name}: ${serverResult.baseUrl}/benchmarks/${benchmark.id}/runs/${runId}`));
3743
+ }
3744
+ }
3745
+ if (process.env.OPENSEARCH_DASHBOARDS_URL) {
3746
+ console.log(chalk3.gray(` OpenSearch Dashboards: ${process.env.OPENSEARCH_DASHBOARDS_URL}`));
3747
+ }
3748
+ if (serverResult.wasStarted && !shouldStopServer) {
3749
+ console.log("");
3750
+ console.log(chalk3.gray(`Server still running on port ${serverConfig.port}`));
3751
+ console.log(chalk3.gray(` Use --stop-server flag to stop after benchmark`));
3752
+ console.log(chalk3.gray(` Or manually: kill $(lsof -t -i:${serverConfig.port})`));
3753
+ }
3754
+ } finally {
3755
+ cleanup();
3756
+ }
3757
+ });
3758
+ return command;
3759
+ }
3760
+
3761
+ // cli/commands/export.ts
3762
+ import { Command as Command4 } from "commander";
3763
+ import chalk4 from "chalk";
3764
+ import { writeFileSync as writeFileSync2 } from "fs";
3765
+
3766
+ // lib/benchmarkExport.ts
3767
+ function generateExportFilename(benchmarkName) {
3768
+ const sanitized = (benchmarkName || "benchmark-export").trim().toLowerCase().replace(/[^a-z0-9_\-\s]/g, "").replace(/\s+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "");
3769
+ return `${sanitized || "benchmark-export"}.json`;
3770
+ }
3771
+
3772
+ // cli/commands/export.ts
3773
+ function createExportCommand() {
3774
+ const command = new Command4("export").description("Export benchmark test cases as JSON").requiredOption("-b, --benchmark <id-or-name>", "Benchmark ID or name").option("-o, --output <file>", "Output file path (default: <benchmark-name>.json)").option("--stdout", "Write to stdout instead of file").action(async (options) => {
3775
+ const config = await loadConfig();
3776
+ const serverResult = await ensureServer(config.server);
3777
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
3778
+ try {
3779
+ const client = new ApiClient(serverResult.baseUrl);
3780
+ const benchmark = await client.findBenchmark(options.benchmark);
3781
+ if (!benchmark) {
3782
+ console.error(chalk4.red(`
3783
+ Error: Benchmark not found: ${options.benchmark}
3784
+ `));
3785
+ process.exit(1);
3786
+ }
3787
+ const exportData = await client.exportBenchmark(benchmark.id);
3788
+ if (options.stdout) {
3789
+ process.stdout.write(JSON.stringify(exportData, null, 2) + "\n");
3790
+ return;
3791
+ }
3792
+ const outputFile = options.output || generateExportFilename(benchmark.name);
3793
+ writeFileSync2(outputFile, JSON.stringify(exportData, null, 2) + "\n", "utf-8");
3794
+ console.log(chalk4.green(`
3795
+ Exported ${exportData.length} test case(s) to ${chalk4.bold(outputFile)}`));
3796
+ console.log(chalk4.gray(` Benchmark: ${benchmark.name} (${benchmark.id})
3797
+ `));
3798
+ } catch (error) {
3799
+ console.error(chalk4.red(`
3800
+ Error: ${error.message}`));
3801
+ console.log(chalk4.gray(" Is the server running? Start with: npm run dev:server\n"));
3802
+ process.exit(1);
3803
+ } finally {
3804
+ cleanup();
3805
+ }
3806
+ });
3807
+ return command;
3808
+ }
3809
+
3810
+ // cli/commands/report.ts
3811
+ import { Command as Command5 } from "commander";
3812
+ import chalk5 from "chalk";
3813
+ import ora3 from "ora";
3814
+ import { writeFileSync as writeFileSync3 } from "fs";
3815
+ function createReportCommand() {
3816
+ const command = new Command5("report").description("Generate a report for a benchmark").requiredOption("-b, --benchmark <id>", "Benchmark name or ID").option("-r, --runs <ids>", "Comma-separated run IDs (default: all runs)").option("-f, --format <type>", "Report format: json, html, pdf", "html").option("-o, --output <file>", "Output file path (auto-generates filename if omitted)").option("--stdout", "Write to stdout (JSON format only)").action(async (options) => {
3817
+ const config = await loadConfig();
3818
+ const serverConfig = { ...DEFAULT_SERVER_CONFIG, ...config.server };
3819
+ const connectSpinner = ora3("Connecting to server...").start();
3820
+ let serverResult;
3821
+ let cleanup;
3822
+ try {
3823
+ serverResult = await ensureServer(serverConfig);
3824
+ cleanup = createServerCleanup(serverResult, false);
3825
+ if (serverResult.wasStarted) {
3826
+ connectSpinner.succeed(`Started server on port ${serverConfig.port}`);
3827
+ } else {
3828
+ connectSpinner.succeed(`Connected to existing server on port ${serverConfig.port}`);
3829
+ }
3830
+ } catch (error) {
3831
+ connectSpinner.fail(
3832
+ `Failed to connect to server: ${error instanceof Error ? error.message : error}`
3833
+ );
3834
+ process.exit(1);
3835
+ }
3836
+ const api = new ApiClient(serverResult.baseUrl);
3837
+ try {
3838
+ const spinner = ora3("Finding benchmark...").start();
3839
+ const benchmark = await api.findBenchmark(options.benchmark);
3840
+ if (!benchmark) {
3841
+ spinner.fail(`Benchmark not found: "${options.benchmark}"`);
3842
+ console.log("");
3843
+ console.log(chalk5.cyan(" Available benchmarks:"));
3844
+ console.log(chalk5.gray(" npx agent-health list benchmarks"));
3845
+ console.log("");
3846
+ process.exit(1);
3847
+ }
3848
+ spinner.succeed(`Found benchmark: ${benchmark.name} (${benchmark.id})`);
3849
+ const params = new URLSearchParams({ format: options.format });
3850
+ if (options.runs) {
3851
+ params.set("runIds", options.runs);
3852
+ }
3853
+ const reportSpinner = ora3(`Generating ${options.format.toUpperCase()} report...`).start();
3854
+ const url = `${serverResult.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmark.id)}/report?${params.toString()}`;
3855
+ const response = await fetch(url);
3856
+ if (!response.ok) {
3857
+ const errorBody = await response.json().catch(() => ({ error: "Unknown error" }));
3858
+ reportSpinner.fail(`Report generation failed: ${errorBody.error}`);
3859
+ process.exit(1);
3860
+ }
3861
+ const contentDisposition = response.headers.get("content-disposition") || "";
3862
+ const filenameMatch = contentDisposition.match(/filename="([^"]+)"/);
3863
+ const defaultFilename = filenameMatch?.[1] || `report.${options.format}`;
3864
+ if (options.stdout) {
3865
+ const text = await response.text();
3866
+ reportSpinner.stop();
3867
+ process.stdout.write(text);
3868
+ } else {
3869
+ const outputPath = options.output || defaultFilename;
3870
+ const contentType = response.headers.get("content-type") || "";
3871
+ if (contentType.includes("application/pdf")) {
3872
+ const buffer = Buffer.from(await response.arrayBuffer());
3873
+ writeFileSync3(outputPath, buffer);
3874
+ } else {
3875
+ const text = await response.text();
3876
+ writeFileSync3(outputPath, text);
3877
+ }
3878
+ reportSpinner.succeed(`Report saved to: ${outputPath}`);
3879
+ }
3880
+ } finally {
3881
+ cleanup();
3882
+ }
3883
+ });
3884
+ return command;
3885
+ }
3886
+
3887
+ // cli/commands/doctor.ts
3888
+ import { Command as Command6 } from "commander";
3889
+ import chalk6 from "chalk";
3890
+ import { existsSync as existsSync3 } from "fs";
3891
+ import { resolve as resolve2 } from "path";
3892
+ function checkConfigFile() {
3893
+ const configInfo = getConfigFileInfo();
3894
+ if (configInfo) {
3895
+ return {
3896
+ name: "Config File",
3897
+ status: "ok",
3898
+ message: `Found: ${configInfo.path.split("/").pop()}`,
3899
+ details: [`Format: ${configInfo.format}`]
3900
+ };
3901
+ }
3902
+ return {
3903
+ name: "Config File",
3904
+ status: "ok",
3905
+ message: "Using defaults + environment variables",
3906
+ details: [
3907
+ "Config file is optional (for custom agents)",
3908
+ "Run `agent-health init` to create one if needed"
3909
+ ]
3910
+ };
3911
+ }
3912
+ function checkEnvFile() {
3913
+ const envPath = resolve2(process.cwd(), ".env");
3914
+ if (existsSync3(envPath)) {
3915
+ return {
3916
+ name: "Environment File",
3917
+ status: "ok",
3918
+ message: "Found: .env"
3919
+ };
3920
+ }
3921
+ return {
3922
+ name: "Environment File",
3923
+ status: "ok",
3924
+ message: "Not using .env file",
3925
+ details: ["Env vars can be set in shell, CI/CD, or via --env-file"]
3926
+ };
3927
+ }
3928
+ function checkAWSCredentials() {
3929
+ const profile = process.env.AWS_PROFILE;
3930
+ const accessKey = process.env.AWS_ACCESS_KEY_ID;
3931
+ const region = process.env.AWS_REGION || process.env.AWS_DEFAULT_REGION;
3932
+ const details = [];
3933
+ if (profile) {
3934
+ details.push(`AWS_PROFILE: ${profile}`);
3935
+ }
3936
+ if (accessKey) {
3937
+ details.push(`AWS_ACCESS_KEY_ID: ${accessKey.substring(0, 8)}...`);
3938
+ }
3939
+ if (region) {
3940
+ details.push(`AWS_REGION: ${region}`);
3941
+ }
3942
+ if (profile || accessKey) {
3943
+ return {
3944
+ name: "AWS Credentials",
3945
+ status: "ok",
3946
+ message: profile ? `Profile: ${profile}` : "Using access key",
3947
+ details: details.length > 0 ? details : void 0
3948
+ };
3949
+ }
3950
+ return {
3951
+ name: "AWS Credentials",
3952
+ status: "warning",
3953
+ message: "No AWS credentials detected",
3954
+ details: [
3955
+ "Set AWS_PROFILE or AWS_ACCESS_KEY_ID for Bedrock judge",
3956
+ "Claude Code connector also requires AWS credentials"
3957
+ ]
3958
+ };
3959
+ }
3960
+ async function checkClaudeCodeCLI() {
3961
+ const { execSync: execSync2 } = await import("child_process");
3962
+ try {
3963
+ execSync2("which claude", { stdio: "pipe" });
3964
+ return {
3965
+ name: "Claude Code CLI",
3966
+ status: "ok",
3967
+ message: "claude command available"
3968
+ };
3969
+ } catch {
3970
+ return {
3971
+ name: "Claude Code CLI",
3972
+ status: "warning",
3973
+ message: "claude command not found",
3974
+ details: [
3975
+ "Install Claude Code: npm install -g @anthropic/claude-code",
3976
+ "Or skip if not using claude-code connector"
3977
+ ]
3978
+ };
3979
+ }
3980
+ }
3981
+ function checkAgents(config) {
3982
+ const agents = config.agents;
3983
+ const details = [];
3984
+ for (const agent of agents) {
3985
+ const connectorType = agent.connectorType || "agui-streaming";
3986
+ const hasConnector = connectorRegistry.get(connectorType);
3987
+ const status = hasConnector ? "\u2713" : "\u2717";
3988
+ details.push(`${status} ${agent.name} (${connectorType})`);
3989
+ }
3990
+ return {
3991
+ name: "Agents",
3992
+ status: "ok",
3993
+ message: `${agents.length} agents configured`,
3994
+ details
3995
+ };
3996
+ }
3997
+ function checkConnectors() {
3998
+ const types = connectorRegistry.getRegisteredTypes();
3999
+ return {
4000
+ name: "Connectors",
4001
+ status: "ok",
4002
+ message: `${types.length} connectors registered`,
4003
+ details: types.map((t) => `\u2713 ${t}`)
4004
+ };
4005
+ }
4006
+ function checkOpenSearchStorage() {
4007
+ const endpoint = process.env.OPENSEARCH_STORAGE_ENDPOINT;
4008
+ const user = process.env.OPENSEARCH_STORAGE_USERNAME;
4009
+ if (endpoint) {
4010
+ return {
4011
+ name: "OpenSearch Storage",
4012
+ status: "ok",
4013
+ message: `Configured: ${endpoint.substring(0, 50)}...`,
4014
+ details: user ? [`Username: ${user}`] : void 0
4015
+ };
4016
+ }
4017
+ return {
4018
+ name: "OpenSearch Storage",
4019
+ status: "warning",
4020
+ message: "Not configured (results won't persist)",
4021
+ details: [
4022
+ "Set OPENSEARCH_STORAGE_ENDPOINT, _USERNAME, _PASSWORD to save results",
4023
+ "Without storage, results are shown in terminal only"
4024
+ ]
4025
+ };
4026
+ }
4027
+ function checkOpenSearchObservability() {
4028
+ const endpoint = process.env.OPENSEARCH_LOGS_ENDPOINT;
4029
+ const user = process.env.OPENSEARCH_LOGS_USERNAME;
4030
+ if (endpoint) {
4031
+ return {
4032
+ name: "OpenSearch Observability",
4033
+ status: "ok",
4034
+ message: `Configured: ${endpoint.substring(0, 50)}...`,
4035
+ details: user ? [`Username: ${user}`] : void 0
4036
+ };
4037
+ }
4038
+ return {
4039
+ name: "OpenSearch Observability",
4040
+ status: "ok",
4041
+ message: "Not configured (optional)",
4042
+ details: [
4043
+ "Set OPENSEARCH_LOGS_ENDPOINT for agent traces",
4044
+ "Only needed for ML-Commons agent observability"
4045
+ ]
4046
+ };
4047
+ }
4048
+ function displayResults(results) {
4049
+ console.log(chalk6.bold("\n Configuration Check\n"));
4050
+ for (const result of results) {
4051
+ const icon = {
4052
+ ok: chalk6.green("\u2713"),
4053
+ warning: chalk6.yellow("\u26A0"),
4054
+ error: chalk6.red("\u2717")
4055
+ }[result.status];
4056
+ const messageColor = {
4057
+ ok: chalk6.green,
4058
+ warning: chalk6.yellow,
4059
+ error: chalk6.red
4060
+ }[result.status];
4061
+ console.log(` ${icon} ${chalk6.bold(result.name)}: ${messageColor(result.message)}`);
4062
+ if (result.details) {
4063
+ for (const detail of result.details) {
4064
+ console.log(chalk6.gray(` ${detail}`));
4065
+ }
4066
+ }
4067
+ }
4068
+ console.log("");
4069
+ const errors = results.filter((r) => r.status === "error").length;
4070
+ const warnings = results.filter((r) => r.status === "warning").length;
4071
+ if (errors > 0) {
4072
+ console.log(chalk6.red(` ${errors} error(s) found. Fix these before running evaluations.
4073
+ `));
4074
+ } else if (warnings > 0) {
4075
+ console.log(chalk6.yellow(` ${warnings} warning(s). Some features may be limited.
4076
+ `));
4077
+ } else {
4078
+ console.log(chalk6.green(" All checks passed!\n"));
4079
+ }
4080
+ }
4081
+ function createDoctorCommand() {
4082
+ const command = new Command6("doctor").description("Check configuration and system requirements").option("-o, --output <format>", "Output format: text, json", "text").action(async (options) => {
4083
+ const results = [];
4084
+ const config = await loadConfig();
4085
+ for (const connector of config.connectors) {
4086
+ connectorRegistry.register(connector);
4087
+ }
4088
+ results.push(checkConfigFile());
4089
+ results.push(checkEnvFile());
4090
+ results.push(checkAWSCredentials());
4091
+ results.push(await checkClaudeCodeCLI());
4092
+ results.push(checkAgents(config));
4093
+ results.push(checkConnectors());
4094
+ results.push(checkOpenSearchStorage());
4095
+ results.push(checkOpenSearchObservability());
4096
+ if (options.output === "json") {
4097
+ console.log(JSON.stringify(results, null, 2));
4098
+ } else {
4099
+ displayResults(results);
4100
+ }
4101
+ });
4102
+ return command;
4103
+ }
4104
+
4105
+ // cli/commands/init.ts
4106
+ import { Command as Command7 } from "commander";
4107
+ import chalk7 from "chalk";
4108
+ import { writeFileSync as writeFileSync4, existsSync as existsSync4 } from "fs";
4109
+ import { resolve as resolve3 } from "path";
4110
+ var TYPESCRIPT_CONFIG = `/*
4111
+ * Agent Health Configuration
4112
+ * See docs/CONFIGURATION.md for full options
4113
+ */
4114
+
4115
+ import { defineConfig, AGUIStreamingConnector, ClaudeCodeConnector } from '@opensearch-project/agent-health';
4116
+
4117
+ export default defineConfig({
4118
+ agents: [
4119
+ // ML-Commons agent (AG-UI streaming)
4120
+ {
4121
+ name: 'ml-commons',
4122
+ key: 'ml-commons',
4123
+ connector: new AGUIStreamingConnector(),
4124
+ endpoint: process.env.MLCOMMONS_ENDPOINT || 'https://localhost:9200/_plugins/_ml/agents/YOUR_AGENT_ID/_execute',
4125
+ auth: {
4126
+ type: 'basic',
4127
+ username: process.env.OPENSEARCH_USER || 'admin',
4128
+ password: process.env.OPENSEARCH_PASS || 'admin',
4129
+ },
4130
+ models: ['claude-sonnet'],
4131
+ },
4132
+
4133
+ // Claude Code CLI agent (optional)
4134
+ // Uncomment to enable Claude Code comparison
4135
+ /*
4136
+ {
4137
+ name: 'claude-code',
4138
+ key: 'claude-code',
4139
+ connector: new ClaudeCodeConnector({
4140
+ env: {
4141
+ AWS_PROFILE: process.env.AWS_PROFILE || 'Bedrock',
4142
+ CLAUDE_CODE_USE_BEDROCK: '1',
4143
+ AWS_REGION: process.env.AWS_REGION || 'us-west-2',
4144
+ },
4145
+ }),
4146
+ endpoint: 'claude', // Command name
4147
+ models: ['claude-sonnet-4'],
4148
+ },
4149
+ */
4150
+ ],
4151
+
4152
+ // Test cases can be inline or loaded from files
4153
+ testCases: './test-cases/*.yaml',
4154
+
4155
+ // Output reporters
4156
+ reporters: [
4157
+ ['console'],
4158
+ ['json', { output: 'report.json' }],
4159
+ ],
4160
+
4161
+ // Judge configuration
4162
+ judge: {
4163
+ provider: 'bedrock',
4164
+ model: 'claude-sonnet',
4165
+ region: process.env.AWS_REGION || 'us-west-2',
4166
+ },
4167
+ });
4168
+ `;
4169
+ var ENV_TEMPLATE = `# Agent Health Environment Configuration
4170
+ # Copy this to .env and fill in your values
4171
+
4172
+ # ============ AWS Configuration ============
4173
+ # For Bedrock judge and Claude Code CLI
4174
+ AWS_PROFILE=Bedrock
4175
+ AWS_REGION=us-west-2
4176
+ # Or use explicit credentials:
4177
+ # AWS_ACCESS_KEY_ID=
4178
+ # AWS_SECRET_ACCESS_KEY=
4179
+ # AWS_SESSION_TOKEN=
4180
+
4181
+ # ============ OpenSearch/ML-Commons ============
4182
+ # Agent endpoint
4183
+ MLCOMMONS_ENDPOINT=https://localhost:9200/_plugins/_ml/agents/YOUR_AGENT_ID/_execute
4184
+
4185
+ # Storage cluster (for test cases, benchmarks persistence)
4186
+ OPENSEARCH_STORAGE_URL=https://localhost:9200
4187
+ OPENSEARCH_STORAGE_USER=admin
4188
+ OPENSEARCH_STORAGE_PASS=admin
4189
+
4190
+ # Optional: Headers for ML-Commons agent data source access
4191
+ # MLCOMMONS_HEADER_OPENSEARCH_URL=
4192
+ # MLCOMMONS_HEADER_AUTHORIZATION=
4193
+
4194
+ # ============ Server Configuration ============
4195
+ # Backend port (default: 4001)
4196
+ # BACKEND_PORT=4001
4197
+ `;
4198
+ var SAMPLE_TEST_CASE = `# Sample Test Case
4199
+ # Place in test-cases/ directory
4200
+
4201
+ id: sample-rca-001
4202
+ name: Sample RCA Test Case
4203
+ version: 1
4204
+
4205
+ labels:
4206
+ - category:RCA
4207
+ - difficulty:Medium
4208
+
4209
+ initialPrompt: |
4210
+ A customer reports that their web application is experiencing
4211
+ slow response times. The issue started approximately 2 hours ago.
4212
+ Please investigate and identify the root cause.
4213
+
4214
+ context:
4215
+ - description: Application logs
4216
+ value: |
4217
+ 2024-01-15 10:00:00 ERROR: Connection timeout to database
4218
+ 2024-01-15 10:00:05 WARN: Retry attempt 1 for DB connection
4219
+ 2024-01-15 10:00:10 ERROR: Connection timeout to database
4220
+
4221
+ expectedOutcomes:
4222
+ - The agent should identify database connectivity issues
4223
+ - The agent should check database server health
4224
+ - The agent should suggest investigating network connectivity
4225
+ `;
4226
+ function createInitCommand() {
4227
+ const command = new Command7("init").description("Initialize configuration files").option("--force", "Overwrite existing files").option("--with-examples", "Include example test case").action(async (options) => {
4228
+ console.log(chalk7.bold("\n Agent Health - Initialize Configuration\n"));
4229
+ const cwd = process.cwd();
4230
+ const files = [];
4231
+ files.push({
4232
+ path: resolve3(cwd, "agent-health.config.ts"),
4233
+ content: TYPESCRIPT_CONFIG,
4234
+ name: "agent-health.config.ts"
4235
+ });
4236
+ files.push({
4237
+ path: resolve3(cwd, ".env.example"),
4238
+ content: ENV_TEMPLATE,
4239
+ name: ".env.example"
4240
+ });
4241
+ if (options.withExamples) {
4242
+ const { mkdirSync } = await import("fs");
4243
+ const testCasesDir = resolve3(cwd, "test-cases");
4244
+ if (!existsSync4(testCasesDir)) {
4245
+ mkdirSync(testCasesDir, { recursive: true });
4246
+ }
4247
+ files.push({
4248
+ path: resolve3(testCasesDir, "sample-rca.yaml"),
4249
+ content: SAMPLE_TEST_CASE,
4250
+ name: "test-cases/sample-rca.yaml"
4251
+ });
4252
+ }
4253
+ let created = 0;
4254
+ let skipped = 0;
4255
+ for (const file of files) {
4256
+ if (existsSync4(file.path) && !options.force) {
4257
+ console.log(chalk7.yellow(` \u26A0 Skipped: ${file.name} (already exists, use --force to overwrite)`));
4258
+ skipped++;
4259
+ } else {
4260
+ writeFileSync4(file.path, file.content);
4261
+ console.log(chalk7.green(` \u2713 Created: ${file.name}`));
4262
+ created++;
4263
+ }
4264
+ }
4265
+ console.log("");
4266
+ if (created > 0) {
4267
+ console.log(chalk7.gray(" Next steps:"));
4268
+ console.log(chalk7.gray(" 1. Copy .env.example to .env and fill in your values"));
4269
+ console.log(chalk7.gray(" 2. Update the config file with your agent endpoint"));
4270
+ console.log(chalk7.gray(" 3. Run `agent-health doctor` to verify configuration"));
4271
+ console.log(chalk7.gray(" 4. Run `agent-health run -t sample-rca-001` to test\n"));
4272
+ }
4273
+ if (skipped > 0) {
4274
+ console.log(chalk7.yellow(` ${skipped} file(s) skipped. Use --force to overwrite.
4275
+ `));
4276
+ }
4277
+ });
4278
+ return command;
4279
+ }
4280
+
4281
+ // cli/commands/migrate.ts
4282
+ import { Command as Command8 } from "commander";
4283
+ import chalk8 from "chalk";
4284
+ import ora4 from "ora";
4285
+ function computeStatsFromReports(run, reports) {
4286
+ const reportsMap = new Map(reports.map((r) => [r.id, r]));
4287
+ let passed = 0;
4288
+ let failed = 0;
4289
+ let pending = 0;
4290
+ const total = Object.keys(run.results || {}).length;
4291
+ Object.values(run.results || {}).forEach((result) => {
4292
+ if (result.status === "pending" || result.status === "running") {
4293
+ pending++;
4294
+ return;
4295
+ }
4296
+ if (result.status === "failed" || result.status === "cancelled") {
4297
+ failed++;
4298
+ return;
4299
+ }
4300
+ if (result.status === "completed" && result.reportId) {
4301
+ const report = reportsMap.get(result.reportId);
4302
+ if (!report) {
4303
+ pending++;
4304
+ return;
4305
+ }
4306
+ if (report.metricsStatus === "pending" || report.metricsStatus === "calculating") {
4307
+ pending++;
4308
+ return;
4309
+ }
4310
+ if (report.passFailStatus === "passed") {
4311
+ passed++;
4312
+ } else {
4313
+ failed++;
4314
+ }
4315
+ } else {
4316
+ pending++;
4317
+ }
4318
+ });
4319
+ return { passed, failed, pending, total };
4320
+ }
4321
+ function createMigrateCommand() {
4322
+ const command = new Command8("migrate").description("One-time migration to add stats to existing benchmark runs").option("--dry-run", "Show what would be migrated without making changes").option("-v, --verbose", "Show detailed progress").action(async (options) => {
4323
+ console.log(chalk8.cyan.bold("\n Benchmark Stats Migration\n"));
4324
+ const config = await loadConfig();
4325
+ const serverResult = await ensureServer(config.server);
4326
+ const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
4327
+ try {
4328
+ const client = new ApiClient(serverResult.baseUrl);
4329
+ const spinner = ora4("Fetching benchmarks...").start();
4330
+ const benchmarks = await client.listBenchmarks();
4331
+ spinner.succeed(`Found ${benchmarks.length} benchmarks`);
4332
+ const migratable = benchmarks.filter(
4333
+ (b) => !b.id.startsWith("demo-") && (b.runs?.length ?? 0) > 0
4334
+ );
4335
+ if (migratable.length === 0) {
4336
+ console.log(chalk8.yellow("\n No benchmarks to migrate.\n"));
4337
+ console.log(chalk8.gray(" Only user-created benchmarks with runs can be migrated."));
4338
+ console.log(chalk8.gray(" Sample data (demo-*) already has stats computed.\n"));
4339
+ return;
4340
+ }
4341
+ console.log(chalk8.gray(`
4342
+ Migrating ${migratable.length} benchmarks with runs...
4343
+ `));
4344
+ let totalRuns = 0;
4345
+ let migratedRuns = 0;
4346
+ let skippedRuns = 0;
4347
+ let errors = 0;
4348
+ for (const benchmark of migratable) {
4349
+ const runs = benchmark.runs || [];
4350
+ totalRuns += runs.length;
4351
+ if (options.verbose) {
4352
+ console.log(chalk8.gray(` Processing: ${benchmark.name} (${runs.length} runs)`));
4353
+ }
4354
+ for (const run of runs) {
4355
+ if (run.stats && typeof run.stats.passed === "number") {
4356
+ skippedRuns++;
4357
+ if (options.verbose) {
4358
+ console.log(chalk8.gray(` \u2713 ${run.name} - already has stats`));
4359
+ }
4360
+ continue;
4361
+ }
4362
+ try {
4363
+ const reportsRes = await fetch(
4364
+ `${serverResult.baseUrl}/api/storage/runs/by-benchmark-run/${benchmark.id}/${run.id}`
4365
+ );
4366
+ if (!reportsRes.ok) {
4367
+ throw new Error(`Failed to fetch reports: ${reportsRes.status}`);
4368
+ }
4369
+ const { runs: reports } = await reportsRes.json();
4370
+ const stats = computeStatsFromReports(run, reports || []);
4371
+ if (options.verbose) {
4372
+ console.log(chalk8.gray(
4373
+ ` \u2192 ${run.name}: passed=${stats.passed}, failed=${stats.failed}, pending=${stats.pending}`
4374
+ ));
4375
+ }
4376
+ if (!options.dryRun) {
4377
+ const updateRes = await fetch(
4378
+ `${serverResult.baseUrl}/api/storage/benchmarks/${benchmark.id}/runs/${run.id}/stats`,
4379
+ {
4380
+ method: "PATCH",
4381
+ headers: { "Content-Type": "application/json" },
4382
+ body: JSON.stringify(stats)
4383
+ }
4384
+ );
4385
+ if (!updateRes.ok) {
4386
+ const errorBody = await updateRes.text();
4387
+ throw new Error(`Failed to update stats: ${errorBody}`);
4388
+ }
4389
+ }
4390
+ migratedRuns++;
4391
+ } catch (error) {
4392
+ errors++;
4393
+ const msg = error instanceof Error ? error.message : "Unknown error";
4394
+ if (options.verbose) {
4395
+ console.log(chalk8.red(` \u2717 ${run.name} - ${msg}`));
4396
+ }
4397
+ }
4398
+ }
4399
+ console.log(
4400
+ options.dryRun ? chalk8.blue(` [DRY RUN] ${benchmark.name} - ${runs.length} runs would be processed`) : chalk8.green(` \u2713 ${benchmark.name} - ${runs.length} runs`)
4401
+ );
4402
+ }
4403
+ console.log(chalk8.bold("\n Migration Summary\n"));
4404
+ console.log(chalk8.gray(` Total runs: ${totalRuns}`));
4405
+ console.log(chalk8.green(` Migrated: ${migratedRuns}`));
4406
+ console.log(chalk8.yellow(` Already done: ${skippedRuns}`));
4407
+ if (errors > 0) {
4408
+ console.log(chalk8.red(` Errors: ${errors}`));
4409
+ }
4410
+ if (options.dryRun) {
4411
+ console.log(chalk8.blue("\n This was a dry run. No changes were made."));
4412
+ console.log(chalk8.blue(" Run without --dry-run to apply changes.\n"));
4413
+ } else {
4414
+ console.log(chalk8.green("\n Migration complete!\n"));
4415
+ }
4416
+ } catch (error) {
4417
+ const msg = error instanceof Error ? error.message : "Unknown error";
4418
+ console.error(chalk8.red(`
4419
+ Error: ${msg}
4420
+ `));
4421
+ process.exit(1);
4422
+ } finally {
4423
+ cleanup();
4424
+ }
4425
+ });
4426
+ return command;
4427
+ }
4428
+
4429
+ // cli/index.ts
4430
+ var __filename3 = fileURLToPath3(import.meta.url);
4431
+ var __dirname3 = dirname3(__filename3);
4432
+ var packageJsonPath2 = join4(__dirname3, "..", "..", "package.json");
4433
+ var version = "0.1.0";
4434
+ try {
4435
+ const packageJson = JSON.parse(readFileSync3(packageJsonPath2, "utf-8"));
4436
+ version = packageJson.version;
4437
+ } catch {
4438
+ }
4439
+ function loadEnvFile(envPath) {
4440
+ const absolutePath = resolve4(process.cwd(), envPath);
4441
+ if (!existsSync5(absolutePath)) {
4442
+ console.error(chalk9.red(`
4443
+ Error: Environment file not found: ${absolutePath}
4444
+ `));
4445
+ process.exit(1);
4446
+ }
4447
+ const result = loadDotenv({ path: absolutePath });
4448
+ if (result.error) {
4449
+ console.error(chalk9.red(`
4450
+ Error loading environment file: ${result.error.message}
4451
+ `));
4452
+ process.exit(1);
4453
+ }
4454
+ console.log(chalk9.gray(` Loaded environment from: ${envPath}`));
4455
+ }
4456
+ var defaultEnvPath = resolve4(process.cwd(), ".env");
4457
+ if (existsSync5(defaultEnvPath)) {
4458
+ loadDotenv({ path: defaultEnvPath });
4459
+ }
4460
+ var program = new Command9();
4461
+ program.name("agent-health").description("Agent Health Evaluation Framework - Evaluate and monitor AI agent performance").version(version).enablePositionalOptions().passThroughOptions();
4462
+ program.option("-p, --port <number>", "Server port", "4001").option("-e, --env-file <path>", "Load environment variables from file (e.g., .env)").option("--no-browser", "Do not open browser automatically");
4463
+ program.action(async (options) => {
4464
+ console.log(chalk9.cyan.bold(`
4465
+ Agent Health v${version} - AI Agent Evaluation Framework
4466
+ `));
4467
+ console.log(chalk9.gray(` Working directory: ${process.cwd()}`));
4468
+ console.log(chalk9.gray(` Package directory: ${__dirname3}`));
4469
+ if (options.envFile) {
4470
+ loadEnvFile(options.envFile);
4471
+ } else if (existsSync5(defaultEnvPath)) {
4472
+ console.log(chalk9.gray(" Auto-loaded .env from current directory"));
4473
+ }
4474
+ const port = parseInt(options.port, 10);
4475
+ const spinner = ora5("Starting server...").start();
4476
+ try {
4477
+ await startServer({ port });
4478
+ spinner.succeed("Server started");
4479
+ console.log(chalk9.gray("\n Configuration:"));
4480
+ console.log(chalk9.gray(` Storage: Sample data (configure OpenSearch for persistence)`));
4481
+ console.log(chalk9.gray(` Agent: Select in UI (Demo Agent for mock, real agents require endpoints)`));
4482
+ console.log(chalk9.gray(` Judge: Select in UI (Demo Judge for mock, Bedrock requires AWS creds)
4483
+ `));
4484
+ const url = `http://localhost:${port}`;
4485
+ console.log(chalk9.green(` Server running at ${chalk9.bold(url)}
4486
+ `));
4487
+ if (options.browser !== false) {
4488
+ console.log(chalk9.gray(" Opening browser..."));
4489
+ await open(url);
4490
+ }
4491
+ console.log(chalk9.gray(" Press Ctrl+C to stop\n"));
4492
+ } catch (error) {
4493
+ spinner.fail("Failed to start server");
4494
+ console.error(chalk9.red(`
4495
+ Error: ${error instanceof Error ? error.message : error}
4496
+ `));
4497
+ process.exit(1);
4498
+ }
4499
+ });
4500
+ program.addCommand(createListCommand());
4501
+ program.addCommand(createRunCommand());
4502
+ program.addCommand(createBenchmarkCommand());
4503
+ program.addCommand(createExportCommand());
4504
+ program.addCommand(createReportCommand());
4505
+ program.addCommand(createDoctorCommand());
4506
+ program.addCommand(createInitCommand());
4507
+ program.addCommand(createMigrateCommand());
4508
+ program.command("serve").description("Start the Agent Health server (same as default action)").option("-p, --port <number>", "Server port", "4001").option("--no-browser", "Do not open browser automatically").action(async (options) => {
4509
+ console.log(chalk9.cyan.bold(`
4510
+ Agent Health v${version} - AI Agent Evaluation Framework
4511
+ `));
4512
+ const port = parseInt(options.port, 10);
4513
+ const spinner = ora5("Starting server...").start();
4514
+ try {
4515
+ await startServer({ port });
4516
+ spinner.succeed("Server started");
4517
+ const url = `http://localhost:${port}`;
4518
+ console.log(chalk9.green(` Server running at ${chalk9.bold(url)}
4519
+ `));
4520
+ if (options.browser !== false) {
4521
+ console.log(chalk9.gray(" Opening browser..."));
4522
+ await open(url);
4523
+ }
4524
+ console.log(chalk9.gray(" Press Ctrl+C to stop\n"));
4525
+ } catch (error) {
4526
+ spinner.fail("Failed to start server");
4527
+ console.error(chalk9.red(`
4528
+ Error: ${error instanceof Error ? error.message : error}
4529
+ `));
4530
+ process.exit(1);
4531
+ }
4532
+ });
4533
+ program.on("command:*", (operands) => {
4534
+ const unknownCommand = operands[0];
4535
+ const availableCommands = program.commands.map((cmd) => cmd.name());
4536
+ console.error(chalk9.red(`
4537
+ Error: Unknown command '${unknownCommand}'`));
4538
+ console.log("");
4539
+ console.log(chalk9.cyan(" Available commands:"));
4540
+ for (const cmd of availableCommands) {
4541
+ console.log(chalk9.gray(` - ${cmd}`));
4542
+ }
4543
+ console.log("");
4544
+ console.log(chalk9.gray(` Run ${chalk9.cyan("agent-health --help")} for usage information.
4545
+ `));
4546
+ process.exitCode = 1;
4547
+ });
4548
+ program.parse();