@opensearch-project/agent-health 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/cli/dist/index.js CHANGED
@@ -1,14 +1,14 @@
1
1
  #!/usr/bin/env node
2
2
 
3
3
  // cli/index.ts
4
- import { Command as Command8 } from "commander";
5
- import chalk8 from "chalk";
4
+ import { Command as Command9 } from "commander";
5
+ import chalk9 from "chalk";
6
6
  import { fileURLToPath as fileURLToPath3 } from "url";
7
7
  import { dirname as dirname3, join as join4, resolve as resolve4 } from "path";
8
8
  import { readFileSync as readFileSync3, existsSync as existsSync5 } from "fs";
9
9
  import { config as loadDotenv } from "dotenv";
10
10
  import open from "open";
11
- import ora4 from "ora";
11
+ import ora5 from "ora";
12
12
 
13
13
  // cli/utils/startServer.ts
14
14
  import { fileURLToPath } from "url";
@@ -49,25 +49,6 @@ import { existsSync as existsSync2 } from "fs";
49
49
  import { resolve } from "path";
50
50
  import { pathToFileURL } from "url";
51
51
 
52
- // lib/debug.ts
53
- var isBrowser = typeof window !== "undefined";
54
- var serverDebugEnabled = !isBrowser && typeof process !== "undefined" && process.env?.DEBUG === "true";
55
- function isDebugEnabled() {
56
- if (isBrowser) {
57
- try {
58
- return localStorage.getItem("agenteval_debug") === "true";
59
- } catch {
60
- return false;
61
- }
62
- }
63
- return serverDebugEnabled;
64
- }
65
- function debug(module, ...args) {
66
- if (isDebugEnabled()) {
67
- console.debug(`[${module}]`, ...args);
68
- }
69
- }
70
-
71
52
  // lib/config.ts
72
53
  var isServerSide = typeof window === "undefined";
73
54
  var SERVER_PORT = isServerSide ? process.env?.VITE_BACKEND_PORT || process.env?.PORT || "4001" : "4001";
@@ -103,10 +84,8 @@ var ENV_CONFIG = {
103
84
  openSearchLogsPassword: getEnvVar("OPENSEARCH_LOGS_PASSWORD", ""),
104
85
  openSearchLogsTracesIndex: getEnvVar("OPENSEARCH_LOGS_TRACES_INDEX", "otel-v1-apm-span-*"),
105
86
  openSearchLogsIndex: getEnvVar("OPENSEARCH_LOGS_INDEX", "ml-commons-logs-*"),
106
- // Per-agent endpoints
107
- langgraphEndpoint: getEnvVar("LANGGRAPH_ENDPOINT", "http://localhost:3000"),
87
+ // ML-Commons agent endpoint
108
88
  mlcommonsEndpoint: getEnvVar("MLCOMMONS_ENDPOINT", "http://localhost:9200/_plugins/_ml/agents/{agent_id}/_execute/stream"),
109
- holmesGptEndpoint: getEnvVar("HOLMESGPT_ENDPOINT", "http://localhost:5050/api/agui/chat"),
110
89
  // ML-Commons agent headers
111
90
  mlcommonsHeaderOpenSearchUrl: getEnvVar("MLCOMMONS_HEADER_OPENSEARCH_URL", ""),
112
91
  mlcommonsHeaderAuthorization: getEnvVar("MLCOMMONS_HEADER_AUTHORIZATION", ""),
@@ -115,6 +94,11 @@ var ENV_CONFIG = {
115
94
  mlcommonsHeaderAwsAccessKeyId: getEnvVar("MLCOMMONS_HEADER_AWS_ACCESS_KEY_ID", ""),
116
95
  mlcommonsHeaderAwsSecretAccessKey: getEnvVar("MLCOMMONS_HEADER_AWS_SECRET_ACCESS_KEY", ""),
117
96
  mlcommonsHeaderAwsSessionToken: getEnvVar("MLCOMMONS_HEADER_AWS_SESSION_TOKEN", ""),
97
+ // Travel Planner multi-agent endpoint (OTel Demo in Docker)
98
+ travelPlannerEndpoint: getEnvVar("TRAVEL_PLANNER_ENDPOINT", "http://localhost:3000"),
99
+ // LiteLLM (optional - for OpenAI-compatible judge/agent endpoints)
100
+ litellmApiKey: getEnvVar("LITELLM_API_KEY", ""),
101
+ litellmEndpoint: getEnvVar("LITELLM_ENDPOINT", "http://localhost:4000/v1/chat/completions"),
118
102
  // Claude Code Telemetry (optional - for OTEL traces from Claude Code)
119
103
  claudeCodeTelemetryEnabled: getEnvVar("CLAUDE_CODE_TELEMETRY_ENABLED", "false") === "true",
120
104
  otelExporterEndpoint: getEnvVar("OTEL_EXPORTER_OTLP_ENDPOINT", ""),
@@ -123,7 +107,6 @@ var ENV_CONFIG = {
123
107
  otelExporterHeaders: getEnvVar("OTEL_EXPORTER_OTLP_HEADERS", "")
124
108
  };
125
109
  function buildMLCommonsHeaders() {
126
- debug("Config", "Building ML-Commons headers");
127
110
  const headers = {};
128
111
  if (ENV_CONFIG.mlcommonsHeaderOpenSearchUrl) {
129
112
  headers["opensearch-url"] = ENV_CONFIG.mlcommonsHeaderOpenSearchUrl;
@@ -147,7 +130,6 @@ function buildMLCommonsHeaders() {
147
130
  headers["aws-session-token"] = ENV_CONFIG.mlcommonsHeaderAwsSessionToken;
148
131
  }
149
132
  }
150
- debug("Config", "ML-Commons headers built, keys:", Object.keys(headers));
151
133
  return headers;
152
134
  }
153
135
 
@@ -187,45 +169,23 @@ var DEFAULT_CONFIG = {
187
169
  headers: {},
188
170
  useTraces: false
189
171
  },
190
- {
191
- key: "langgraph",
192
- name: "Langgraph",
193
- endpoint: ENV_CONFIG.langgraphEndpoint,
194
- description: "Langgraph AG-UI agent server",
195
- connectorType: "agui-streaming",
196
- models: [
197
- "claude-sonnet-4.5",
198
- "claude-sonnet-4",
199
- "claude-haiku-3.5"
200
- ],
201
- headers: {},
202
- useTraces: true
203
- },
204
172
  {
205
173
  key: "mlcommons-local",
206
174
  name: "ML-Commons (Localhost)",
207
175
  endpoint: ENV_CONFIG.mlcommonsEndpoint,
208
176
  description: "Local OpenSearch ML-Commons conversational agent",
209
177
  connectorType: "agui-streaming",
210
- models: [
211
- "claude-sonnet-4.5",
212
- "claude-sonnet-4",
213
- "claude-haiku-3.5"
214
- ],
178
+ models: ["claude-sonnet-4.5", "claude-sonnet-4", "claude-haiku-3.5"],
215
179
  headers: buildMLCommonsHeaders(),
216
180
  useTraces: true
217
181
  },
218
182
  {
219
- key: "holmesgpt",
220
- name: "HolmesGPT",
221
- endpoint: ENV_CONFIG.holmesGptEndpoint,
222
- description: "HolmesGPT AI-powered RCA agent (AG-UI)",
183
+ key: "travel-planner",
184
+ name: "Travel Planner",
185
+ endpoint: ENV_CONFIG.travelPlannerEndpoint,
186
+ description: "Multi-agent Travel Planner demo (requires OTel Demo running via Docker)",
223
187
  connectorType: "agui-streaming",
224
- models: [
225
- "claude-sonnet-4.5",
226
- "claude-sonnet-4",
227
- "claude-haiku-3.5"
228
- ],
188
+ models: ["claude-sonnet-4.5", "claude-sonnet-4", "claude-haiku-3.5"],
229
189
  headers: {},
230
190
  useTraces: true
231
191
  },
@@ -233,20 +193,12 @@ var DEFAULT_CONFIG = {
233
193
  key: "claude-code",
234
194
  name: "Claude Code",
235
195
  endpoint: "claude",
236
- // Command name, not URL
237
196
  description: "Claude Code CLI agent (requires claude command installed)",
238
197
  connectorType: "claude-code",
239
198
  models: ["claude-sonnet-4"],
240
199
  headers: {},
241
- get useTraces() {
242
- return !!(ENV_CONFIG.claudeCodeTelemetryEnabled && ENV_CONFIG.otelExporterEndpoint);
243
- },
244
- // connectorConfig env vars are evaluated at runtime by getClaudeCodeConnectorEnv()
245
- get connectorConfig() {
246
- return {
247
- env: getClaudeCodeConnectorEnv()
248
- };
249
- }
200
+ useTraces: ENV_CONFIG.claudeCodeTelemetryEnabled && !!ENV_CONFIG.otelExporterEndpoint,
201
+ connectorConfig: { env: getClaudeCodeConnectorEnv() }
250
202
  }
251
203
  ],
252
204
  models: {
@@ -277,6 +229,13 @@ var DEFAULT_CONFIG = {
277
229
  provider: "bedrock",
278
230
  context_window: 2e5,
279
231
  max_output_tokens: 4096
232
+ },
233
+ "gpt-4o": {
234
+ model_id: "gpt-4o",
235
+ display_name: "GPT-4o (via LiteLLM)",
236
+ provider: "litellm",
237
+ context_window: 128e3,
238
+ max_output_tokens: 4096
280
239
  }
281
240
  },
282
241
  defaults: {
@@ -501,6 +460,44 @@ var ConnectorRegistryImpl = class {
501
460
  };
502
461
  var connectorRegistry = new ConnectorRegistryImpl();
503
462
 
463
+ // lib/debug.ts
464
+ import fs from "fs";
465
+ import path from "path";
466
+ var isBrowser = typeof window !== "undefined";
467
+ var CONFIG_FILENAME = "agent-health.config.json";
468
+ var serverDebugEnabled = false;
469
+ if (!isBrowser) {
470
+ try {
471
+ const configPath = path.join(process.cwd(), CONFIG_FILENAME);
472
+ if (fs.existsSync(configPath)) {
473
+ const content = fs.readFileSync(configPath, "utf-8");
474
+ const config = JSON.parse(content) || {};
475
+ serverDebugEnabled = config.debug === true;
476
+ } else if (process.env?.DEBUG === "true") {
477
+ serverDebugEnabled = true;
478
+ }
479
+ } catch (err) {
480
+ if (process.env?.DEBUG === "true") {
481
+ serverDebugEnabled = true;
482
+ }
483
+ }
484
+ }
485
+ function isDebugEnabled() {
486
+ if (isBrowser) {
487
+ try {
488
+ return localStorage.getItem("agenteval_debug") === "true";
489
+ } catch {
490
+ return false;
491
+ }
492
+ }
493
+ return serverDebugEnabled;
494
+ }
495
+ function debug(module, ...args) {
496
+ if (isDebugEnabled()) {
497
+ console.debug(`[${module}]`, ...args);
498
+ }
499
+ }
500
+
504
501
  // services/connectors/base/BaseConnector.ts
505
502
  var BaseConnector = class {
506
503
  /**
@@ -631,9 +628,11 @@ var SSEClient = class {
631
628
  this.abortController = new AbortController();
632
629
  debug("SSE", "Connecting to", url);
633
630
  debug("SSE", "Method:", method);
634
- debug("SSE", "Payload:", JSON.stringify(body, null, 2).substring(0, 500));
631
+ debug("SSE", "Headers:", headers);
632
+ debug("SSE", "Payload:", body ? JSON.stringify(body, null, 2).substring(0, 500) : "none");
633
+ debug("SSE", "Timeout:", idleTimeoutMs, "ms");
635
634
  try {
636
- const response = await fetch(url, {
635
+ const requestConfig = {
637
636
  method,
638
637
  headers: {
639
638
  "Content-Type": "application/json",
@@ -642,17 +641,28 @@ var SSEClient = class {
642
641
  },
643
642
  body: body ? JSON.stringify(body) : void 0,
644
643
  signal: this.abortController.signal
645
- });
644
+ };
645
+ debug("SSE", "Request config:", JSON.stringify(requestConfig, null, 2).substring(0, 500));
646
+ const response = await fetch(url, requestConfig);
647
+ debug("SSE", "Response received:", response.status, response.statusText);
646
648
  if (!response.ok) {
647
- throw new Error(`HTTP ${response.status}: ${response.statusText}`);
649
+ let errorBody = "";
650
+ try {
651
+ errorBody = await response.text();
652
+ debug("SSE", "Error response body:", errorBody.substring(0, 500));
653
+ } catch {
654
+ debug("SSE", "Could not read error response body");
655
+ }
656
+ throw new Error(`HTTP ${response.status}: ${response.statusText}${errorBody ? ` - ${errorBody}` : ""}`);
648
657
  }
649
658
  if (!response.body) {
650
659
  throw new Error("Response body is null");
651
660
  }
652
- debug("SSE", "Connected, streaming events...");
661
+ console.info("[SSE] Connected to agent endpoint, streaming events...");
653
662
  debug("SSE", "Response status:", response.status);
654
663
  debug("SSE", "Content-Type:", response.headers.get("content-type"));
655
664
  const completionReason = await this.processStream(response.body, onEvent, completeOnRunEnd, idleTimeoutMs);
665
+ console.info(`[SSE] Stream completed: ${completionReason}`);
656
666
  debug("SSE", `Stream completed: ${completionReason}`);
657
667
  onComplete?.();
658
668
  } catch (error) {
@@ -662,9 +672,36 @@ var SSEClient = class {
662
672
  onComplete?.();
663
673
  } else {
664
674
  console.error("[SSE] Stream error:", error.message);
675
+ console.error("[SSE] Debug mode is:", isDebugEnabled() ? "ENABLED \u2705" : "DISABLED \u274C");
676
+ const errorDetails = {
677
+ name: error.name,
678
+ message: error.message,
679
+ stack: error.stack,
680
+ cause: error.cause,
681
+ url,
682
+ method,
683
+ headers
684
+ };
685
+ debug("SSE", "Error details:", errorDetails);
686
+ if (error.message.includes("fetch failed")) {
687
+ debug("SSE", '\u{1F4A1} Diagnostic: "fetch failed" typically means:');
688
+ debug("SSE", " - Connection refused (endpoint not running)");
689
+ debug("SSE", " - DNS resolution failed (invalid hostname)");
690
+ debug("SSE", " - Network unreachable (firewall/VPN issues)");
691
+ debug("SSE", " - SSL/TLS certificate issues (self-signed cert)");
692
+ debug("SSE", ` Check if ${url} is accessible`);
693
+ } else if (error.message.includes("timeout")) {
694
+ debug("SSE", "\u{1F4A1} Diagnostic: Request timed out - endpoint may be slow or unresponsive");
695
+ } else if (error.message.includes("ENOTFOUND")) {
696
+ debug("SSE", "\u{1F4A1} Diagnostic: DNS lookup failed - hostname not found");
697
+ } else if (error.message.includes("ECONNREFUSED")) {
698
+ debug("SSE", "\u{1F4A1} Diagnostic: Connection refused - service not listening on this port");
699
+ }
665
700
  onError?.(error);
666
701
  }
667
702
  } else {
703
+ console.error("[SSE] Unknown error:", error);
704
+ debug("SSE", "Unknown error details:", error);
668
705
  onError?.(new Error("Unknown error occurred"));
669
706
  }
670
707
  }
@@ -708,8 +745,11 @@ var SSEClient = class {
708
745
  debug("SSE", "Raw event:", data.substring(0, 200) + (data.length > 200 ? "..." : ""));
709
746
  try {
710
747
  const event = JSON.parse(data);
711
- debug("SSE", "Parsed:", event.type);
748
+ debug("SSE", "Parsed event:", event.type);
712
749
  eventCount++;
750
+ if (eventCount % 10 === 0) {
751
+ console.info(`[SSE] Processed ${eventCount} events...`);
752
+ }
713
753
  onEvent(event);
714
754
  if (completeOnRunEnd && (event.type === AGUIEventType.RUN_FINISHED || event.type === AGUIEventType.RUN_ERROR)) {
715
755
  debug("SSE", `Received ${event.type}, completing stream`);
@@ -1520,10 +1560,123 @@ var RESTConnector = class extends BaseConnector {
1520
1560
  };
1521
1561
  var restConnector = new RESTConnector();
1522
1562
 
1563
+ // services/connectors/litellm/LiteLLMConnector.ts
1564
+ var LiteLLMConnector = class extends BaseConnector {
1565
+ constructor() {
1566
+ super(...arguments);
1567
+ this.type = "litellm";
1568
+ this.name = "LiteLLM / OpenAI-compatible";
1569
+ this.supportsStreaming = false;
1570
+ }
1571
+ /**
1572
+ * Build OpenAI Chat Completion payload from test case
1573
+ */
1574
+ buildPayload(request) {
1575
+ const messages = [];
1576
+ if (request.testCase.context && request.testCase.context.length > 0) {
1577
+ const contextText = request.testCase.context.map((c) => typeof c === "string" ? c : JSON.stringify(c)).join("\n");
1578
+ messages.push({
1579
+ role: "system",
1580
+ content: contextText
1581
+ });
1582
+ }
1583
+ messages.push({
1584
+ role: "user",
1585
+ content: request.testCase.initialPrompt
1586
+ });
1587
+ const payload = {
1588
+ model: request.modelId,
1589
+ messages
1590
+ };
1591
+ if (request.testCase.tools && request.testCase.tools.length > 0) {
1592
+ payload.tools = request.testCase.tools.map((tool) => ({
1593
+ type: "function",
1594
+ function: {
1595
+ name: tool.name,
1596
+ description: tool.description || "",
1597
+ parameters: tool.parameters || {}
1598
+ }
1599
+ }));
1600
+ }
1601
+ return payload;
1602
+ }
1603
+ /**
1604
+ * Execute OpenAI-compatible Chat Completion request
1605
+ */
1606
+ async execute(endpoint, request, auth, onProgress, onRawEvent) {
1607
+ const payload = request.payload || this.buildPayload(request);
1608
+ const headers = this.buildAuthHeaders(auth);
1609
+ this.debug("Executing LiteLLM request");
1610
+ this.debug("Endpoint:", endpoint);
1611
+ this.debug("Model:", payload.model);
1612
+ const response = await fetch(endpoint, {
1613
+ method: "POST",
1614
+ headers: {
1615
+ "Content-Type": "application/json",
1616
+ ...headers
1617
+ },
1618
+ body: JSON.stringify(payload)
1619
+ });
1620
+ if (!response.ok) {
1621
+ const errorText = await response.text();
1622
+ throw new Error(`LiteLLM request failed: ${response.status} - ${errorText}`);
1623
+ }
1624
+ const data = await response.json();
1625
+ onRawEvent?.(data);
1626
+ const trajectory = this.parseResponse(data);
1627
+ trajectory.forEach((step) => onProgress?.(step));
1628
+ return {
1629
+ trajectory,
1630
+ runId: data.id || null,
1631
+ rawEvents: [data],
1632
+ metadata: {
1633
+ model: data.model,
1634
+ usage: data.usage,
1635
+ finishReason: data.choices?.[0]?.finish_reason
1636
+ }
1637
+ };
1638
+ }
1639
+ /**
1640
+ * Parse OpenAI Chat Completion response into trajectory steps
1641
+ */
1642
+ parseResponse(data) {
1643
+ const steps = [];
1644
+ const choice = data.choices?.[0];
1645
+ if (!choice) {
1646
+ steps.push(this.createStep("response", JSON.stringify(data, null, 2)));
1647
+ return steps;
1648
+ }
1649
+ const message = choice.message;
1650
+ if (message.tool_calls && message.tool_calls.length > 0) {
1651
+ for (const toolCall of message.tool_calls) {
1652
+ let toolArgs;
1653
+ try {
1654
+ toolArgs = JSON.parse(toolCall.function.arguments);
1655
+ } catch {
1656
+ toolArgs = toolCall.function.arguments;
1657
+ }
1658
+ steps.push(this.createStep("action", `Calling ${toolCall.function.name}...`, {
1659
+ toolName: toolCall.function.name,
1660
+ toolArgs
1661
+ }));
1662
+ }
1663
+ }
1664
+ if (message.content) {
1665
+ steps.push(this.createStep("response", message.content));
1666
+ }
1667
+ if (steps.length === 0) {
1668
+ steps.push(this.createStep("response", "(empty response)"));
1669
+ }
1670
+ return steps;
1671
+ }
1672
+ };
1673
+ var litellmConnector = new LiteLLMConnector();
1674
+
1523
1675
  // services/connectors/index.ts
1524
1676
  connectorRegistry.register(aguiStreamingConnector);
1525
1677
  connectorRegistry.register(mockConnector);
1526
1678
  connectorRegistry.register(restConnector);
1679
+ connectorRegistry.register(litellmConnector);
1527
1680
  console.log("[Connectors] Browser-safe connectors registered:", connectorRegistry.getRegisteredTypes().join(", "));
1528
1681
 
1529
1682
  // services/connectors/subprocess/SubprocessConnector.ts
@@ -3359,26 +3512,50 @@ function displaySummaryTable(allResults, totalTestCases) {
3359
3512
  console.log(chalk3.bold("Benchmark Summary"));
3360
3513
  console.log(table.toString());
3361
3514
  }
3362
- function exportResults(benchmark, allResults, exportPath) {
3363
- const exportData = {
3364
- benchmark: {
3365
- id: benchmark.id,
3366
- name: benchmark.name,
3367
- testCaseCount: benchmark.testCaseIds.length
3368
- },
3369
- runs: allResults.map((r) => ({
3370
- agent: { key: r.agent.key, name: r.agent.name },
3371
- runId: r.run?.id || r.runId,
3372
- status: r.run?.status,
3373
- passed: r.passed,
3374
- failed: r.failed,
3375
- passRate: benchmark.testCaseIds.length > 0 ? r.passed / benchmark.testCaseIds.length * 100 : 0,
3376
- results: r.run?.results,
3377
- reports: r.reports
3378
- })),
3379
- exportedAt: (/* @__PURE__ */ new Date()).toISOString()
3380
- };
3381
- writeFileSync(exportPath, JSON.stringify(exportData, null, 2));
3515
+ async function exportResults(benchmark, allResults, exportPath, format, serverBaseUrl) {
3516
+ if (format !== "json") {
3517
+ const runIds = allResults.map((r) => r.run?.id || r.runId).filter((id) => !!id);
3518
+ const params = new URLSearchParams({ format });
3519
+ if (runIds.length > 0) {
3520
+ params.set("runIds", runIds.join(","));
3521
+ }
3522
+ const url = `${serverBaseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmark.id)}/report?${params.toString()}`;
3523
+ const response = await fetch(url);
3524
+ if (!response.ok) {
3525
+ const errorBody = await response.json().catch(() => ({ error: "Unknown error" }));
3526
+ console.error(chalk3.red(`
3527
+ Export failed: ${errorBody.error}`));
3528
+ return;
3529
+ }
3530
+ const contentType = response.headers.get("content-type") || "";
3531
+ if (contentType.includes("application/pdf")) {
3532
+ const buffer = Buffer.from(await response.arrayBuffer());
3533
+ writeFileSync(exportPath, buffer);
3534
+ } else {
3535
+ const text = await response.text();
3536
+ writeFileSync(exportPath, text);
3537
+ }
3538
+ } else {
3539
+ const exportData = {
3540
+ benchmark: {
3541
+ id: benchmark.id,
3542
+ name: benchmark.name,
3543
+ testCaseCount: benchmark.testCaseIds.length
3544
+ },
3545
+ runs: allResults.map((r) => ({
3546
+ agent: { key: r.agent.key, name: r.agent.name },
3547
+ runId: r.run?.id || r.runId,
3548
+ status: r.run?.status,
3549
+ passed: r.passed,
3550
+ failed: r.failed,
3551
+ passRate: benchmark.testCaseIds.length > 0 ? r.passed / benchmark.testCaseIds.length * 100 : 0,
3552
+ results: r.run?.results,
3553
+ reports: r.reports
3554
+ })),
3555
+ exportedAt: (/* @__PURE__ */ new Date()).toISOString()
3556
+ };
3557
+ writeFileSync(exportPath, JSON.stringify(exportData, null, 2));
3558
+ }
3382
3559
  console.log(chalk3.green(`
3383
3560
  Results exported to: ${exportPath}`));
3384
3561
  }
@@ -3388,7 +3565,7 @@ function createBenchmarkCommand() {
3388
3565
  "Agent key (can be specified multiple times)",
3389
3566
  (val, arr) => [...arr, val],
3390
3567
  []
3391
- ).option("-m, --model <id>", "Model ID (uses agent default if not specified)").option("-o, --output <format>", "Output format: table, json", "table").option("--export <path>", "Export results to JSON file").option("-v, --verbose", "Show detailed output").option("--stop-server", "Stop the server after benchmark completes (default: keep running)").action(async (options) => {
3568
+ ).option("-m, --model <id>", "Model ID (uses agent default if not specified)").option("-o, --output <format>", "Output format: table, json", "table").option("--export <path>", "Export results to file").option("--format <type>", "Report format for --export: json (default), html, pdf", "json").option("-v, --verbose", "Show detailed output").option("--stop-server", "Stop the server after benchmark completes (default: keep running)").action(async (options) => {
3392
3569
  console.log(chalk3.bold("\nAgent Health - Benchmark Runner\n"));
3393
3570
  const config = await loadConfig();
3394
3571
  const serverConfig = { ...DEFAULT_SERVER_CONFIG, ...config.server };
@@ -3555,7 +3732,7 @@ function createBenchmarkCommand() {
3555
3732
  displaySummaryTable(allResults, benchmark.testCaseIds.length);
3556
3733
  }
3557
3734
  if (options.export) {
3558
- exportResults(benchmark, allResults, options.export);
3735
+ await exportResults(benchmark, allResults, options.export, options.format, serverResult.baseUrl);
3559
3736
  }
3560
3737
  console.log("");
3561
3738
  console.log(chalk3.cyan("View results:"));
@@ -3630,9 +3807,86 @@ function createExportCommand() {
3630
3807
  return command;
3631
3808
  }
3632
3809
 
3633
- // cli/commands/doctor.ts
3810
+ // cli/commands/report.ts
3634
3811
  import { Command as Command5 } from "commander";
3635
3812
  import chalk5 from "chalk";
3813
+ import ora3 from "ora";
3814
+ import { writeFileSync as writeFileSync3 } from "fs";
3815
+ function createReportCommand() {
3816
+ const command = new Command5("report").description("Generate a report for a benchmark").requiredOption("-b, --benchmark <id>", "Benchmark name or ID").option("-r, --runs <ids>", "Comma-separated run IDs (default: all runs)").option("-f, --format <type>", "Report format: json, html, pdf", "html").option("-o, --output <file>", "Output file path (auto-generates filename if omitted)").option("--stdout", "Write to stdout (JSON format only)").action(async (options) => {
3817
+ const config = await loadConfig();
3818
+ const serverConfig = { ...DEFAULT_SERVER_CONFIG, ...config.server };
3819
+ const connectSpinner = ora3("Connecting to server...").start();
3820
+ let serverResult;
3821
+ let cleanup;
3822
+ try {
3823
+ serverResult = await ensureServer(serverConfig);
3824
+ cleanup = createServerCleanup(serverResult, false);
3825
+ if (serverResult.wasStarted) {
3826
+ connectSpinner.succeed(`Started server on port ${serverConfig.port}`);
3827
+ } else {
3828
+ connectSpinner.succeed(`Connected to existing server on port ${serverConfig.port}`);
3829
+ }
3830
+ } catch (error) {
3831
+ connectSpinner.fail(
3832
+ `Failed to connect to server: ${error instanceof Error ? error.message : error}`
3833
+ );
3834
+ process.exit(1);
3835
+ }
3836
+ const api = new ApiClient(serverResult.baseUrl);
3837
+ try {
3838
+ const spinner = ora3("Finding benchmark...").start();
3839
+ const benchmark = await api.findBenchmark(options.benchmark);
3840
+ if (!benchmark) {
3841
+ spinner.fail(`Benchmark not found: "${options.benchmark}"`);
3842
+ console.log("");
3843
+ console.log(chalk5.cyan(" Available benchmarks:"));
3844
+ console.log(chalk5.gray(" npx agent-health list benchmarks"));
3845
+ console.log("");
3846
+ process.exit(1);
3847
+ }
3848
+ spinner.succeed(`Found benchmark: ${benchmark.name} (${benchmark.id})`);
3849
+ const params = new URLSearchParams({ format: options.format });
3850
+ if (options.runs) {
3851
+ params.set("runIds", options.runs);
3852
+ }
3853
+ const reportSpinner = ora3(`Generating ${options.format.toUpperCase()} report...`).start();
3854
+ const url = `${serverResult.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmark.id)}/report?${params.toString()}`;
3855
+ const response = await fetch(url);
3856
+ if (!response.ok) {
3857
+ const errorBody = await response.json().catch(() => ({ error: "Unknown error" }));
3858
+ reportSpinner.fail(`Report generation failed: ${errorBody.error}`);
3859
+ process.exit(1);
3860
+ }
3861
+ const contentDisposition = response.headers.get("content-disposition") || "";
3862
+ const filenameMatch = contentDisposition.match(/filename="([^"]+)"/);
3863
+ const defaultFilename = filenameMatch?.[1] || `report.${options.format}`;
3864
+ if (options.stdout) {
3865
+ const text = await response.text();
3866
+ reportSpinner.stop();
3867
+ process.stdout.write(text);
3868
+ } else {
3869
+ const outputPath = options.output || defaultFilename;
3870
+ const contentType = response.headers.get("content-type") || "";
3871
+ if (contentType.includes("application/pdf")) {
3872
+ const buffer = Buffer.from(await response.arrayBuffer());
3873
+ writeFileSync3(outputPath, buffer);
3874
+ } else {
3875
+ const text = await response.text();
3876
+ writeFileSync3(outputPath, text);
3877
+ }
3878
+ reportSpinner.succeed(`Report saved to: ${outputPath}`);
3879
+ }
3880
+ } finally {
3881
+ cleanup();
3882
+ }
3883
+ });
3884
+ return command;
3885
+ }
3886
+
3887
+ // cli/commands/doctor.ts
3888
+ import { Command as Command6 } from "commander";
3889
+ import chalk6 from "chalk";
3636
3890
  import { existsSync as existsSync3 } from "fs";
3637
3891
  import { resolve as resolve2 } from "path";
3638
3892
  function checkConfigFile() {
@@ -3792,22 +4046,22 @@ function checkOpenSearchObservability() {
3792
4046
  };
3793
4047
  }
3794
4048
  function displayResults(results) {
3795
- console.log(chalk5.bold("\n Configuration Check\n"));
4049
+ console.log(chalk6.bold("\n Configuration Check\n"));
3796
4050
  for (const result of results) {
3797
4051
  const icon = {
3798
- ok: chalk5.green("\u2713"),
3799
- warning: chalk5.yellow("\u26A0"),
3800
- error: chalk5.red("\u2717")
4052
+ ok: chalk6.green("\u2713"),
4053
+ warning: chalk6.yellow("\u26A0"),
4054
+ error: chalk6.red("\u2717")
3801
4055
  }[result.status];
3802
4056
  const messageColor = {
3803
- ok: chalk5.green,
3804
- warning: chalk5.yellow,
3805
- error: chalk5.red
4057
+ ok: chalk6.green,
4058
+ warning: chalk6.yellow,
4059
+ error: chalk6.red
3806
4060
  }[result.status];
3807
- console.log(` ${icon} ${chalk5.bold(result.name)}: ${messageColor(result.message)}`);
4061
+ console.log(` ${icon} ${chalk6.bold(result.name)}: ${messageColor(result.message)}`);
3808
4062
  if (result.details) {
3809
4063
  for (const detail of result.details) {
3810
- console.log(chalk5.gray(` ${detail}`));
4064
+ console.log(chalk6.gray(` ${detail}`));
3811
4065
  }
3812
4066
  }
3813
4067
  }
@@ -3815,17 +4069,17 @@ function displayResults(results) {
3815
4069
  const errors = results.filter((r) => r.status === "error").length;
3816
4070
  const warnings = results.filter((r) => r.status === "warning").length;
3817
4071
  if (errors > 0) {
3818
- console.log(chalk5.red(` ${errors} error(s) found. Fix these before running evaluations.
4072
+ console.log(chalk6.red(` ${errors} error(s) found. Fix these before running evaluations.
3819
4073
  `));
3820
4074
  } else if (warnings > 0) {
3821
- console.log(chalk5.yellow(` ${warnings} warning(s). Some features may be limited.
4075
+ console.log(chalk6.yellow(` ${warnings} warning(s). Some features may be limited.
3822
4076
  `));
3823
4077
  } else {
3824
- console.log(chalk5.green(" All checks passed!\n"));
4078
+ console.log(chalk6.green(" All checks passed!\n"));
3825
4079
  }
3826
4080
  }
3827
4081
  function createDoctorCommand() {
3828
- const command = new Command5("doctor").description("Check configuration and system requirements").option("-o, --output <format>", "Output format: text, json", "text").action(async (options) => {
4082
+ const command = new Command6("doctor").description("Check configuration and system requirements").option("-o, --output <format>", "Output format: text, json", "text").action(async (options) => {
3829
4083
  const results = [];
3830
4084
  const config = await loadConfig();
3831
4085
  for (const connector of config.connectors) {
@@ -3849,9 +4103,9 @@ function createDoctorCommand() {
3849
4103
  }
3850
4104
 
3851
4105
  // cli/commands/init.ts
3852
- import { Command as Command6 } from "commander";
3853
- import chalk6 from "chalk";
3854
- import { writeFileSync as writeFileSync3, existsSync as existsSync4 } from "fs";
4106
+ import { Command as Command7 } from "commander";
4107
+ import chalk7 from "chalk";
4108
+ import { writeFileSync as writeFileSync4, existsSync as existsSync4 } from "fs";
3855
4109
  import { resolve as resolve3 } from "path";
3856
4110
  var TYPESCRIPT_CONFIG = `/*
3857
4111
  * Agent Health Configuration
@@ -3970,8 +4224,8 @@ expectedOutcomes:
3970
4224
  - The agent should suggest investigating network connectivity
3971
4225
  `;
3972
4226
  function createInitCommand() {
3973
- const command = new Command6("init").description("Initialize configuration files").option("--force", "Overwrite existing files").option("--with-examples", "Include example test case").action(async (options) => {
3974
- console.log(chalk6.bold("\n Agent Health - Initialize Configuration\n"));
4227
+ const command = new Command7("init").description("Initialize configuration files").option("--force", "Overwrite existing files").option("--with-examples", "Include example test case").action(async (options) => {
4228
+ console.log(chalk7.bold("\n Agent Health - Initialize Configuration\n"));
3975
4229
  const cwd = process.cwd();
3976
4230
  const files = [];
3977
4231
  files.push({
@@ -4000,24 +4254,24 @@ function createInitCommand() {
4000
4254
  let skipped = 0;
4001
4255
  for (const file of files) {
4002
4256
  if (existsSync4(file.path) && !options.force) {
4003
- console.log(chalk6.yellow(` \u26A0 Skipped: ${file.name} (already exists, use --force to overwrite)`));
4257
+ console.log(chalk7.yellow(` \u26A0 Skipped: ${file.name} (already exists, use --force to overwrite)`));
4004
4258
  skipped++;
4005
4259
  } else {
4006
- writeFileSync3(file.path, file.content);
4007
- console.log(chalk6.green(` \u2713 Created: ${file.name}`));
4260
+ writeFileSync4(file.path, file.content);
4261
+ console.log(chalk7.green(` \u2713 Created: ${file.name}`));
4008
4262
  created++;
4009
4263
  }
4010
4264
  }
4011
4265
  console.log("");
4012
4266
  if (created > 0) {
4013
- console.log(chalk6.gray(" Next steps:"));
4014
- console.log(chalk6.gray(" 1. Copy .env.example to .env and fill in your values"));
4015
- console.log(chalk6.gray(" 2. Update the config file with your agent endpoint"));
4016
- console.log(chalk6.gray(" 3. Run `agent-health doctor` to verify configuration"));
4017
- console.log(chalk6.gray(" 4. Run `agent-health run -t sample-rca-001` to test\n"));
4267
+ console.log(chalk7.gray(" Next steps:"));
4268
+ console.log(chalk7.gray(" 1. Copy .env.example to .env and fill in your values"));
4269
+ console.log(chalk7.gray(" 2. Update the config file with your agent endpoint"));
4270
+ console.log(chalk7.gray(" 3. Run `agent-health doctor` to verify configuration"));
4271
+ console.log(chalk7.gray(" 4. Run `agent-health run -t sample-rca-001` to test\n"));
4018
4272
  }
4019
4273
  if (skipped > 0) {
4020
- console.log(chalk6.yellow(` ${skipped} file(s) skipped. Use --force to overwrite.
4274
+ console.log(chalk7.yellow(` ${skipped} file(s) skipped. Use --force to overwrite.
4021
4275
  `));
4022
4276
  }
4023
4277
  });
@@ -4025,9 +4279,9 @@ function createInitCommand() {
4025
4279
  }
4026
4280
 
4027
4281
  // cli/commands/migrate.ts
4028
- import { Command as Command7 } from "commander";
4029
- import chalk7 from "chalk";
4030
- import ora3 from "ora";
4282
+ import { Command as Command8 } from "commander";
4283
+ import chalk8 from "chalk";
4284
+ import ora4 from "ora";
4031
4285
  function computeStatsFromReports(run, reports) {
4032
4286
  const reportsMap = new Map(reports.map((r) => [r.id, r]));
4033
4287
  let passed = 0;
@@ -4065,26 +4319,26 @@ function computeStatsFromReports(run, reports) {
4065
4319
  return { passed, failed, pending, total };
4066
4320
  }
4067
4321
  function createMigrateCommand() {
4068
- const command = new Command7("migrate").description("One-time migration to add stats to existing benchmark runs").option("--dry-run", "Show what would be migrated without making changes").option("-v, --verbose", "Show detailed progress").action(async (options) => {
4069
- console.log(chalk7.cyan.bold("\n Benchmark Stats Migration\n"));
4322
+ const command = new Command8("migrate").description("One-time migration to add stats to existing benchmark runs").option("--dry-run", "Show what would be migrated without making changes").option("-v, --verbose", "Show detailed progress").action(async (options) => {
4323
+ console.log(chalk8.cyan.bold("\n Benchmark Stats Migration\n"));
4070
4324
  const config = await loadConfig();
4071
4325
  const serverResult = await ensureServer(config.server);
4072
4326
  const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
4073
4327
  try {
4074
4328
  const client = new ApiClient(serverResult.baseUrl);
4075
- const spinner = ora3("Fetching benchmarks...").start();
4329
+ const spinner = ora4("Fetching benchmarks...").start();
4076
4330
  const benchmarks = await client.listBenchmarks();
4077
4331
  spinner.succeed(`Found ${benchmarks.length} benchmarks`);
4078
4332
  const migratable = benchmarks.filter(
4079
4333
  (b) => !b.id.startsWith("demo-") && (b.runs?.length ?? 0) > 0
4080
4334
  );
4081
4335
  if (migratable.length === 0) {
4082
- console.log(chalk7.yellow("\n No benchmarks to migrate.\n"));
4083
- console.log(chalk7.gray(" Only user-created benchmarks with runs can be migrated."));
4084
- console.log(chalk7.gray(" Sample data (demo-*) already has stats computed.\n"));
4336
+ console.log(chalk8.yellow("\n No benchmarks to migrate.\n"));
4337
+ console.log(chalk8.gray(" Only user-created benchmarks with runs can be migrated."));
4338
+ console.log(chalk8.gray(" Sample data (demo-*) already has stats computed.\n"));
4085
4339
  return;
4086
4340
  }
4087
- console.log(chalk7.gray(`
4341
+ console.log(chalk8.gray(`
4088
4342
  Migrating ${migratable.length} benchmarks with runs...
4089
4343
  `));
4090
4344
  let totalRuns = 0;
@@ -4095,13 +4349,13 @@ function createMigrateCommand() {
4095
4349
  const runs = benchmark.runs || [];
4096
4350
  totalRuns += runs.length;
4097
4351
  if (options.verbose) {
4098
- console.log(chalk7.gray(` Processing: ${benchmark.name} (${runs.length} runs)`));
4352
+ console.log(chalk8.gray(` Processing: ${benchmark.name} (${runs.length} runs)`));
4099
4353
  }
4100
4354
  for (const run of runs) {
4101
4355
  if (run.stats && typeof run.stats.passed === "number") {
4102
4356
  skippedRuns++;
4103
4357
  if (options.verbose) {
4104
- console.log(chalk7.gray(` \u2713 ${run.name} - already has stats`));
4358
+ console.log(chalk8.gray(` \u2713 ${run.name} - already has stats`));
4105
4359
  }
4106
4360
  continue;
4107
4361
  }
@@ -4115,7 +4369,7 @@ function createMigrateCommand() {
4115
4369
  const { runs: reports } = await reportsRes.json();
4116
4370
  const stats = computeStatsFromReports(run, reports || []);
4117
4371
  if (options.verbose) {
4118
- console.log(chalk7.gray(
4372
+ console.log(chalk8.gray(
4119
4373
  ` \u2192 ${run.name}: passed=${stats.passed}, failed=${stats.failed}, pending=${stats.pending}`
4120
4374
  ));
4121
4375
  }
@@ -4138,30 +4392,30 @@ function createMigrateCommand() {
4138
4392
  errors++;
4139
4393
  const msg = error instanceof Error ? error.message : "Unknown error";
4140
4394
  if (options.verbose) {
4141
- console.log(chalk7.red(` \u2717 ${run.name} - ${msg}`));
4395
+ console.log(chalk8.red(` \u2717 ${run.name} - ${msg}`));
4142
4396
  }
4143
4397
  }
4144
4398
  }
4145
4399
  console.log(
4146
- options.dryRun ? chalk7.blue(` [DRY RUN] ${benchmark.name} - ${runs.length} runs would be processed`) : chalk7.green(` \u2713 ${benchmark.name} - ${runs.length} runs`)
4400
+ options.dryRun ? chalk8.blue(` [DRY RUN] ${benchmark.name} - ${runs.length} runs would be processed`) : chalk8.green(` \u2713 ${benchmark.name} - ${runs.length} runs`)
4147
4401
  );
4148
4402
  }
4149
- console.log(chalk7.bold("\n Migration Summary\n"));
4150
- console.log(chalk7.gray(` Total runs: ${totalRuns}`));
4151
- console.log(chalk7.green(` Migrated: ${migratedRuns}`));
4152
- console.log(chalk7.yellow(` Already done: ${skippedRuns}`));
4403
+ console.log(chalk8.bold("\n Migration Summary\n"));
4404
+ console.log(chalk8.gray(` Total runs: ${totalRuns}`));
4405
+ console.log(chalk8.green(` Migrated: ${migratedRuns}`));
4406
+ console.log(chalk8.yellow(` Already done: ${skippedRuns}`));
4153
4407
  if (errors > 0) {
4154
- console.log(chalk7.red(` Errors: ${errors}`));
4408
+ console.log(chalk8.red(` Errors: ${errors}`));
4155
4409
  }
4156
4410
  if (options.dryRun) {
4157
- console.log(chalk7.blue("\n This was a dry run. No changes were made."));
4158
- console.log(chalk7.blue(" Run without --dry-run to apply changes.\n"));
4411
+ console.log(chalk8.blue("\n This was a dry run. No changes were made."));
4412
+ console.log(chalk8.blue(" Run without --dry-run to apply changes.\n"));
4159
4413
  } else {
4160
- console.log(chalk7.green("\n Migration complete!\n"));
4414
+ console.log(chalk8.green("\n Migration complete!\n"));
4161
4415
  }
4162
4416
  } catch (error) {
4163
4417
  const msg = error instanceof Error ? error.message : "Unknown error";
4164
- console.error(chalk7.red(`
4418
+ console.error(chalk8.red(`
4165
4419
  Error: ${msg}
4166
4420
  `));
4167
4421
  process.exit(1);
@@ -4185,59 +4439,59 @@ try {
4185
4439
  function loadEnvFile(envPath) {
4186
4440
  const absolutePath = resolve4(process.cwd(), envPath);
4187
4441
  if (!existsSync5(absolutePath)) {
4188
- console.error(chalk8.red(`
4442
+ console.error(chalk9.red(`
4189
4443
  Error: Environment file not found: ${absolutePath}
4190
4444
  `));
4191
4445
  process.exit(1);
4192
4446
  }
4193
4447
  const result = loadDotenv({ path: absolutePath });
4194
4448
  if (result.error) {
4195
- console.error(chalk8.red(`
4449
+ console.error(chalk9.red(`
4196
4450
  Error loading environment file: ${result.error.message}
4197
4451
  `));
4198
4452
  process.exit(1);
4199
4453
  }
4200
- console.log(chalk8.gray(` Loaded environment from: ${envPath}`));
4454
+ console.log(chalk9.gray(` Loaded environment from: ${envPath}`));
4201
4455
  }
4202
4456
  var defaultEnvPath = resolve4(process.cwd(), ".env");
4203
4457
  if (existsSync5(defaultEnvPath)) {
4204
4458
  loadDotenv({ path: defaultEnvPath });
4205
4459
  }
4206
- var program = new Command8();
4460
+ var program = new Command9();
4207
4461
  program.name("agent-health").description("Agent Health Evaluation Framework - Evaluate and monitor AI agent performance").version(version).enablePositionalOptions().passThroughOptions();
4208
4462
  program.option("-p, --port <number>", "Server port", "4001").option("-e, --env-file <path>", "Load environment variables from file (e.g., .env)").option("--no-browser", "Do not open browser automatically");
4209
4463
  program.action(async (options) => {
4210
- console.log(chalk8.cyan.bold(`
4464
+ console.log(chalk9.cyan.bold(`
4211
4465
  Agent Health v${version} - AI Agent Evaluation Framework
4212
4466
  `));
4213
- console.log(chalk8.gray(` Working directory: ${process.cwd()}`));
4214
- console.log(chalk8.gray(` Package directory: ${__dirname3}`));
4467
+ console.log(chalk9.gray(` Working directory: ${process.cwd()}`));
4468
+ console.log(chalk9.gray(` Package directory: ${__dirname3}`));
4215
4469
  if (options.envFile) {
4216
4470
  loadEnvFile(options.envFile);
4217
4471
  } else if (existsSync5(defaultEnvPath)) {
4218
- console.log(chalk8.gray(" Auto-loaded .env from current directory"));
4472
+ console.log(chalk9.gray(" Auto-loaded .env from current directory"));
4219
4473
  }
4220
4474
  const port = parseInt(options.port, 10);
4221
- const spinner = ora4("Starting server...").start();
4475
+ const spinner = ora5("Starting server...").start();
4222
4476
  try {
4223
4477
  await startServer({ port });
4224
4478
  spinner.succeed("Server started");
4225
- console.log(chalk8.gray("\n Configuration:"));
4226
- console.log(chalk8.gray(` Storage: Sample data (configure OpenSearch for persistence)`));
4227
- console.log(chalk8.gray(` Agent: Select in UI (Demo Agent for mock, real agents require endpoints)`));
4228
- console.log(chalk8.gray(` Judge: Select in UI (Demo Judge for mock, Bedrock requires AWS creds)
4479
+ console.log(chalk9.gray("\n Configuration:"));
4480
+ console.log(chalk9.gray(` Storage: Sample data (configure OpenSearch for persistence)`));
4481
+ console.log(chalk9.gray(` Agent: Select in UI (Demo Agent for mock, real agents require endpoints)`));
4482
+ console.log(chalk9.gray(` Judge: Select in UI (Demo Judge for mock, Bedrock requires AWS creds)
4229
4483
  `));
4230
4484
  const url = `http://localhost:${port}`;
4231
- console.log(chalk8.green(` Server running at ${chalk8.bold(url)}
4485
+ console.log(chalk9.green(` Server running at ${chalk9.bold(url)}
4232
4486
  `));
4233
4487
  if (options.browser !== false) {
4234
- console.log(chalk8.gray(" Opening browser..."));
4488
+ console.log(chalk9.gray(" Opening browser..."));
4235
4489
  await open(url);
4236
4490
  }
4237
- console.log(chalk8.gray(" Press Ctrl+C to stop\n"));
4491
+ console.log(chalk9.gray(" Press Ctrl+C to stop\n"));
4238
4492
  } catch (error) {
4239
4493
  spinner.fail("Failed to start server");
4240
- console.error(chalk8.red(`
4494
+ console.error(chalk9.red(`
4241
4495
  Error: ${error instanceof Error ? error.message : error}
4242
4496
  `));
4243
4497
  process.exit(1);
@@ -4247,29 +4501,30 @@ program.addCommand(createListCommand());
4247
4501
  program.addCommand(createRunCommand());
4248
4502
  program.addCommand(createBenchmarkCommand());
4249
4503
  program.addCommand(createExportCommand());
4504
+ program.addCommand(createReportCommand());
4250
4505
  program.addCommand(createDoctorCommand());
4251
4506
  program.addCommand(createInitCommand());
4252
4507
  program.addCommand(createMigrateCommand());
4253
4508
  program.command("serve").description("Start the Agent Health server (same as default action)").option("-p, --port <number>", "Server port", "4001").option("--no-browser", "Do not open browser automatically").action(async (options) => {
4254
- console.log(chalk8.cyan.bold(`
4509
+ console.log(chalk9.cyan.bold(`
4255
4510
  Agent Health v${version} - AI Agent Evaluation Framework
4256
4511
  `));
4257
4512
  const port = parseInt(options.port, 10);
4258
- const spinner = ora4("Starting server...").start();
4513
+ const spinner = ora5("Starting server...").start();
4259
4514
  try {
4260
4515
  await startServer({ port });
4261
4516
  spinner.succeed("Server started");
4262
4517
  const url = `http://localhost:${port}`;
4263
- console.log(chalk8.green(` Server running at ${chalk8.bold(url)}
4518
+ console.log(chalk9.green(` Server running at ${chalk9.bold(url)}
4264
4519
  `));
4265
4520
  if (options.browser !== false) {
4266
- console.log(chalk8.gray(" Opening browser..."));
4521
+ console.log(chalk9.gray(" Opening browser..."));
4267
4522
  await open(url);
4268
4523
  }
4269
- console.log(chalk8.gray(" Press Ctrl+C to stop\n"));
4524
+ console.log(chalk9.gray(" Press Ctrl+C to stop\n"));
4270
4525
  } catch (error) {
4271
4526
  spinner.fail("Failed to start server");
4272
- console.error(chalk8.red(`
4527
+ console.error(chalk9.red(`
4273
4528
  Error: ${error instanceof Error ? error.message : error}
4274
4529
  `));
4275
4530
  process.exit(1);
@@ -4278,15 +4533,15 @@ program.command("serve").description("Start the Agent Health server (same as def
4278
4533
  program.on("command:*", (operands) => {
4279
4534
  const unknownCommand = operands[0];
4280
4535
  const availableCommands = program.commands.map((cmd) => cmd.name());
4281
- console.error(chalk8.red(`
4536
+ console.error(chalk9.red(`
4282
4537
  Error: Unknown command '${unknownCommand}'`));
4283
4538
  console.log("");
4284
- console.log(chalk8.cyan(" Available commands:"));
4539
+ console.log(chalk9.cyan(" Available commands:"));
4285
4540
  for (const cmd of availableCommands) {
4286
- console.log(chalk8.gray(` - ${cmd}`));
4541
+ console.log(chalk9.gray(` - ${cmd}`));
4287
4542
  }
4288
4543
  console.log("");
4289
- console.log(chalk8.gray(` Run ${chalk8.cyan("agent-health --help")} for usage information.
4544
+ console.log(chalk9.gray(` Run ${chalk9.cyan("agent-health --help")} for usage information.
4290
4545
  `));
4291
4546
  process.exitCode = 1;
4292
4547
  });