@opensearch-project/agent-health 0.0.1 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.txt +201 -0
- package/README.md +350 -0
- package/bin/cli.js +32 -0
- package/cli/dist/index.js +4548 -0
- package/dist/06363ae0dcb276526e56d844ae57f1eb2e16d1dc.svg +5 -0
- package/dist/17cc03b81ececabf3c8b3c3bf4b8203cd3028613.svg +5 -0
- package/dist/25f48f7350a1cb261d7ab5492972b53a075f1716.svg +5 -0
- package/dist/3d66ad340959f7e4ae00687e9f14adbd930abe01.svg +5 -0
- package/dist/621824624feb2bb70de09132809d9cce164b8bcb.svg +5 -0
- package/dist/81a42dbbd00769e7d823c228f5193c23f6c35e20.svg +5 -0
- package/dist/agent-health-style-guide.html +415 -0
- package/dist/assets/index-D5yuaEp4.js +267 -0
- package/dist/assets/index-D6wGwYUm.css +1 -0
- package/dist/assets/opensearch-logo-DV6ruVFq.svg +5 -0
- package/dist/index.html +22 -0
- package/dist/optimize-loop.svg +48 -0
- package/lib/dist/config/index.js +363 -0
- package/lib/dist/index.js +1589 -0
- package/package.json +148 -11
- package/server/dist/app.js +13933 -0
- package/server/dist/index.js +13958 -0
- package/index.js +0 -13
|
@@ -0,0 +1,4548 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
// cli/index.ts
|
|
4
|
+
import { Command as Command9 } from "commander";
|
|
5
|
+
import chalk9 from "chalk";
|
|
6
|
+
import { fileURLToPath as fileURLToPath3 } from "url";
|
|
7
|
+
import { dirname as dirname3, join as join4, resolve as resolve4 } from "path";
|
|
8
|
+
import { readFileSync as readFileSync3, existsSync as existsSync5 } from "fs";
|
|
9
|
+
import { config as loadDotenv } from "dotenv";
|
|
10
|
+
import open from "open";
|
|
11
|
+
import ora5 from "ora";
|
|
12
|
+
|
|
13
|
+
// cli/utils/startServer.ts
|
|
14
|
+
import { fileURLToPath } from "url";
|
|
15
|
+
import { dirname, join } from "path";
|
|
16
|
+
import { existsSync } from "fs";
|
|
17
|
+
var __filename = fileURLToPath(import.meta.url);
|
|
18
|
+
var __dirname = dirname(__filename);
|
|
19
|
+
function findPackageRoot() {
|
|
20
|
+
let dir = __dirname;
|
|
21
|
+
for (let i = 0; i < 5; i++) {
|
|
22
|
+
if (existsSync(join(dir, "package.json"))) {
|
|
23
|
+
return dir;
|
|
24
|
+
}
|
|
25
|
+
dir = dirname(dir);
|
|
26
|
+
}
|
|
27
|
+
return join(__dirname, "..");
|
|
28
|
+
}
|
|
29
|
+
async function startServer(options) {
|
|
30
|
+
process.env.VITE_BACKEND_PORT = String(options.port);
|
|
31
|
+
const packageRoot = findPackageRoot();
|
|
32
|
+
const serverPath = join(packageRoot, "server", "dist", "app.js");
|
|
33
|
+
const { createApp } = await import(serverPath);
|
|
34
|
+
const app = await createApp();
|
|
35
|
+
return new Promise((resolve5) => {
|
|
36
|
+
app.listen(options.port, "0.0.0.0", () => {
|
|
37
|
+
resolve5();
|
|
38
|
+
});
|
|
39
|
+
});
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// cli/commands/list.ts
|
|
43
|
+
import { Command } from "commander";
|
|
44
|
+
import chalk from "chalk";
|
|
45
|
+
import Table from "cli-table3";
|
|
46
|
+
|
|
47
|
+
// lib/config/loader.ts
|
|
48
|
+
import { existsSync as existsSync2 } from "fs";
|
|
49
|
+
import { resolve } from "path";
|
|
50
|
+
import { pathToFileURL } from "url";
|
|
51
|
+
|
|
52
|
+
// lib/config.ts
|
|
53
|
+
var isServerSide = typeof window === "undefined";
|
|
54
|
+
var SERVER_PORT = isServerSide ? process.env?.VITE_BACKEND_PORT || process.env?.PORT || "4001" : "4001";
|
|
55
|
+
var BACKEND_URL = isServerSide ? `http://localhost:${SERVER_PORT}` : "";
|
|
56
|
+
var browserEnv = {};
|
|
57
|
+
if (typeof window !== "undefined" && typeof document !== "undefined") {
|
|
58
|
+
try {
|
|
59
|
+
browserEnv = import.meta?.env || {};
|
|
60
|
+
} catch {
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
var getEnvVar = (key, defaultValue) => {
|
|
64
|
+
if (isServerSide) {
|
|
65
|
+
return process.env?.[key] || defaultValue || "";
|
|
66
|
+
}
|
|
67
|
+
return browserEnv[key] || defaultValue || "";
|
|
68
|
+
};
|
|
69
|
+
var ENV_CONFIG = {
|
|
70
|
+
// Backend server - empty string means relative URLs
|
|
71
|
+
backendUrl: BACKEND_URL,
|
|
72
|
+
// API endpoints (derived from backend URL)
|
|
73
|
+
judgeApiUrl: `${BACKEND_URL}/api/judge`,
|
|
74
|
+
storageApiUrl: `${BACKEND_URL}/api/storage`,
|
|
75
|
+
agentProxyUrl: `${BACKEND_URL}/api/agent`,
|
|
76
|
+
openSearchProxyUrl: `${BACKEND_URL}/api/opensearch/logs`,
|
|
77
|
+
// AWS/Bedrock
|
|
78
|
+
awsRegion: getEnvVar("AWS_REGION", "us-east-1"),
|
|
79
|
+
awsProfile: getEnvVar("AWS_PROFILE", "default"),
|
|
80
|
+
bedrockModelId: getEnvVar("BEDROCK_MODEL_ID", "anthropic.claude-3-5-sonnet-20241022-v2:0"),
|
|
81
|
+
// OpenSearch Logs (for fetching agent observability data)
|
|
82
|
+
openSearchLogsEndpoint: getEnvVar("OPENSEARCH_LOGS_ENDPOINT", ""),
|
|
83
|
+
openSearchLogsUsername: getEnvVar("OPENSEARCH_LOGS_USERNAME", ""),
|
|
84
|
+
openSearchLogsPassword: getEnvVar("OPENSEARCH_LOGS_PASSWORD", ""),
|
|
85
|
+
openSearchLogsTracesIndex: getEnvVar("OPENSEARCH_LOGS_TRACES_INDEX", "otel-v1-apm-span-*"),
|
|
86
|
+
openSearchLogsIndex: getEnvVar("OPENSEARCH_LOGS_INDEX", "ml-commons-logs-*"),
|
|
87
|
+
// ML-Commons agent endpoint
|
|
88
|
+
mlcommonsEndpoint: getEnvVar("MLCOMMONS_ENDPOINT", "http://localhost:9200/_plugins/_ml/agents/{agent_id}/_execute/stream"),
|
|
89
|
+
// ML-Commons agent headers
|
|
90
|
+
mlcommonsHeaderOpenSearchUrl: getEnvVar("MLCOMMONS_HEADER_OPENSEARCH_URL", ""),
|
|
91
|
+
mlcommonsHeaderAuthorization: getEnvVar("MLCOMMONS_HEADER_AUTHORIZATION", ""),
|
|
92
|
+
mlcommonsHeaderAwsRegion: getEnvVar("MLCOMMONS_HEADER_AWS_REGION", ""),
|
|
93
|
+
mlcommonsHeaderAwsServiceName: getEnvVar("MLCOMMONS_HEADER_AWS_SERVICE_NAME", "es"),
|
|
94
|
+
mlcommonsHeaderAwsAccessKeyId: getEnvVar("MLCOMMONS_HEADER_AWS_ACCESS_KEY_ID", ""),
|
|
95
|
+
mlcommonsHeaderAwsSecretAccessKey: getEnvVar("MLCOMMONS_HEADER_AWS_SECRET_ACCESS_KEY", ""),
|
|
96
|
+
mlcommonsHeaderAwsSessionToken: getEnvVar("MLCOMMONS_HEADER_AWS_SESSION_TOKEN", ""),
|
|
97
|
+
// Travel Planner multi-agent endpoint (OTel Demo in Docker)
|
|
98
|
+
travelPlannerEndpoint: getEnvVar("TRAVEL_PLANNER_ENDPOINT", "http://localhost:3000"),
|
|
99
|
+
// LiteLLM (optional - for OpenAI-compatible judge/agent endpoints)
|
|
100
|
+
litellmApiKey: getEnvVar("LITELLM_API_KEY", ""),
|
|
101
|
+
litellmEndpoint: getEnvVar("LITELLM_ENDPOINT", "http://localhost:4000/v1/chat/completions"),
|
|
102
|
+
// Claude Code Telemetry (optional - for OTEL traces from Claude Code)
|
|
103
|
+
claudeCodeTelemetryEnabled: getEnvVar("CLAUDE_CODE_TELEMETRY_ENABLED", "false") === "true",
|
|
104
|
+
otelExporterEndpoint: getEnvVar("OTEL_EXPORTER_OTLP_ENDPOINT", ""),
|
|
105
|
+
otelServiceName: getEnvVar("OTEL_SERVICE_NAME", "claude-code-agent"),
|
|
106
|
+
otelExporterProtocol: getEnvVar("OTEL_EXPORTER_OTLP_PROTOCOL", ""),
|
|
107
|
+
otelExporterHeaders: getEnvVar("OTEL_EXPORTER_OTLP_HEADERS", "")
|
|
108
|
+
};
|
|
109
|
+
function buildMLCommonsHeaders() {
|
|
110
|
+
const headers = {};
|
|
111
|
+
if (ENV_CONFIG.mlcommonsHeaderOpenSearchUrl) {
|
|
112
|
+
headers["opensearch-url"] = ENV_CONFIG.mlcommonsHeaderOpenSearchUrl;
|
|
113
|
+
}
|
|
114
|
+
if (ENV_CONFIG.mlcommonsHeaderAwsRegion) {
|
|
115
|
+
headers["aws-region"] = ENV_CONFIG.mlcommonsHeaderAwsRegion;
|
|
116
|
+
}
|
|
117
|
+
if (ENV_CONFIG.mlcommonsHeaderAuthorization) {
|
|
118
|
+
headers["Authorization"] = ENV_CONFIG.mlcommonsHeaderAuthorization;
|
|
119
|
+
} else {
|
|
120
|
+
if (ENV_CONFIG.mlcommonsHeaderAwsServiceName) {
|
|
121
|
+
headers["aws-service-name"] = ENV_CONFIG.mlcommonsHeaderAwsServiceName;
|
|
122
|
+
}
|
|
123
|
+
if (ENV_CONFIG.mlcommonsHeaderAwsAccessKeyId) {
|
|
124
|
+
headers["aws-access-key-id"] = ENV_CONFIG.mlcommonsHeaderAwsAccessKeyId;
|
|
125
|
+
}
|
|
126
|
+
if (ENV_CONFIG.mlcommonsHeaderAwsSecretAccessKey) {
|
|
127
|
+
headers["aws-secret-access-key"] = ENV_CONFIG.mlcommonsHeaderAwsSecretAccessKey;
|
|
128
|
+
}
|
|
129
|
+
if (ENV_CONFIG.mlcommonsHeaderAwsSessionToken) {
|
|
130
|
+
headers["aws-session-token"] = ENV_CONFIG.mlcommonsHeaderAwsSessionToken;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
return headers;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// lib/constants.ts
|
|
137
|
+
function getClaudeCodeConnectorEnv() {
|
|
138
|
+
const env = {
|
|
139
|
+
AWS_PROFILE: process.env.AWS_PROFILE || "Bedrock",
|
|
140
|
+
CLAUDE_CODE_USE_BEDROCK: "1",
|
|
141
|
+
AWS_REGION: process.env.AWS_REGION || "us-west-2",
|
|
142
|
+
DISABLE_PROMPT_CACHING: "1",
|
|
143
|
+
DISABLE_ERROR_REPORTING: "1"
|
|
144
|
+
};
|
|
145
|
+
if (ENV_CONFIG.claudeCodeTelemetryEnabled && ENV_CONFIG.otelExporterEndpoint) {
|
|
146
|
+
env.CLAUDE_CODE_ENABLE_TELEMETRY = "1";
|
|
147
|
+
env.OTEL_EXPORTER_OTLP_ENDPOINT = ENV_CONFIG.otelExporterEndpoint;
|
|
148
|
+
env.OTEL_SERVICE_NAME = ENV_CONFIG.otelServiceName;
|
|
149
|
+
if (ENV_CONFIG.otelExporterProtocol) {
|
|
150
|
+
env.OTEL_EXPORTER_OTLP_PROTOCOL = ENV_CONFIG.otelExporterProtocol;
|
|
151
|
+
}
|
|
152
|
+
if (ENV_CONFIG.otelExporterHeaders) {
|
|
153
|
+
env.OTEL_EXPORTER_OTLP_HEADERS = ENV_CONFIG.otelExporterHeaders;
|
|
154
|
+
}
|
|
155
|
+
} else {
|
|
156
|
+
env.DISABLE_TELEMETRY = "1";
|
|
157
|
+
}
|
|
158
|
+
return env;
|
|
159
|
+
}
|
|
160
|
+
var DEFAULT_CONFIG = {
|
|
161
|
+
agents: [
|
|
162
|
+
{
|
|
163
|
+
key: "demo",
|
|
164
|
+
name: "Demo Agent",
|
|
165
|
+
endpoint: "mock://demo",
|
|
166
|
+
description: "Mock agent for testing (simulated responses)",
|
|
167
|
+
connectorType: "mock",
|
|
168
|
+
models: ["demo-model"],
|
|
169
|
+
headers: {},
|
|
170
|
+
useTraces: false
|
|
171
|
+
},
|
|
172
|
+
{
|
|
173
|
+
key: "mlcommons-local",
|
|
174
|
+
name: "ML-Commons (Localhost)",
|
|
175
|
+
endpoint: ENV_CONFIG.mlcommonsEndpoint,
|
|
176
|
+
description: "Local OpenSearch ML-Commons conversational agent",
|
|
177
|
+
connectorType: "agui-streaming",
|
|
178
|
+
models: ["claude-sonnet-4.5", "claude-sonnet-4", "claude-haiku-3.5"],
|
|
179
|
+
headers: buildMLCommonsHeaders(),
|
|
180
|
+
useTraces: true
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
key: "travel-planner",
|
|
184
|
+
name: "Travel Planner",
|
|
185
|
+
endpoint: ENV_CONFIG.travelPlannerEndpoint,
|
|
186
|
+
description: "Multi-agent Travel Planner demo (requires OTel Demo running via Docker)",
|
|
187
|
+
connectorType: "agui-streaming",
|
|
188
|
+
models: ["claude-sonnet-4.5", "claude-sonnet-4", "claude-haiku-3.5"],
|
|
189
|
+
headers: {},
|
|
190
|
+
useTraces: true
|
|
191
|
+
},
|
|
192
|
+
{
|
|
193
|
+
key: "claude-code",
|
|
194
|
+
name: "Claude Code",
|
|
195
|
+
endpoint: "claude",
|
|
196
|
+
description: "Claude Code CLI agent (requires claude command installed)",
|
|
197
|
+
connectorType: "claude-code",
|
|
198
|
+
models: ["claude-sonnet-4"],
|
|
199
|
+
headers: {},
|
|
200
|
+
useTraces: ENV_CONFIG.claudeCodeTelemetryEnabled && !!ENV_CONFIG.otelExporterEndpoint,
|
|
201
|
+
connectorConfig: { env: getClaudeCodeConnectorEnv() }
|
|
202
|
+
}
|
|
203
|
+
],
|
|
204
|
+
models: {
|
|
205
|
+
"demo-model": {
|
|
206
|
+
model_id: "mock://demo-model",
|
|
207
|
+
display_name: "Demo Model",
|
|
208
|
+
provider: "demo",
|
|
209
|
+
context_window: 2e5,
|
|
210
|
+
max_output_tokens: 4096
|
|
211
|
+
},
|
|
212
|
+
"claude-sonnet-4.5": {
|
|
213
|
+
model_id: "us.anthropic.claude-sonnet-4-5-20250929-v1:0",
|
|
214
|
+
display_name: "Claude Sonnet 4.5",
|
|
215
|
+
provider: "bedrock",
|
|
216
|
+
context_window: 2e5,
|
|
217
|
+
max_output_tokens: 4096
|
|
218
|
+
},
|
|
219
|
+
"claude-sonnet-4": {
|
|
220
|
+
model_id: "us.anthropic.claude-sonnet-4-20250514-v1:0",
|
|
221
|
+
display_name: "Claude Sonnet 4",
|
|
222
|
+
provider: "bedrock",
|
|
223
|
+
context_window: 2e5,
|
|
224
|
+
max_output_tokens: 4096
|
|
225
|
+
},
|
|
226
|
+
"claude-haiku-3.5": {
|
|
227
|
+
model_id: "us.anthropic.claude-3-5-haiku-20241022-v1:0",
|
|
228
|
+
display_name: "Claude Haiku 3.5",
|
|
229
|
+
provider: "bedrock",
|
|
230
|
+
context_window: 2e5,
|
|
231
|
+
max_output_tokens: 4096
|
|
232
|
+
},
|
|
233
|
+
"gpt-4o": {
|
|
234
|
+
model_id: "gpt-4o",
|
|
235
|
+
display_name: "GPT-4o (via LiteLLM)",
|
|
236
|
+
provider: "litellm",
|
|
237
|
+
context_window: 128e3,
|
|
238
|
+
max_output_tokens: 4096
|
|
239
|
+
}
|
|
240
|
+
},
|
|
241
|
+
defaults: {
|
|
242
|
+
retry_attempts: 2,
|
|
243
|
+
retry_delay_ms: 1e3
|
|
244
|
+
}
|
|
245
|
+
};
|
|
246
|
+
|
|
247
|
+
// lib/config/loader.ts
|
|
248
|
+
var DEFAULT_SERVER_CONFIG = {
|
|
249
|
+
port: 4001,
|
|
250
|
+
reuseExistingServer: !process.env.CI,
|
|
251
|
+
startTimeout: 3e4
|
|
252
|
+
};
|
|
253
|
+
var CONFIG_FILE_NAMES = [
|
|
254
|
+
"agent-health.config.ts",
|
|
255
|
+
"agent-health.config.js",
|
|
256
|
+
"agent-health.config.mjs"
|
|
257
|
+
];
|
|
258
|
+
function findConfigFile(cwd = process.cwd()) {
|
|
259
|
+
for (const fileName of CONFIG_FILE_NAMES) {
|
|
260
|
+
const filePath = resolve(cwd, fileName);
|
|
261
|
+
if (existsSync2(filePath)) {
|
|
262
|
+
const format = fileName.endsWith(".ts") ? "typescript" : "javascript";
|
|
263
|
+
return { path: filePath, format, exists: true };
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
return null;
|
|
267
|
+
}
|
|
268
|
+
function toAgentConfig(userAgent) {
|
|
269
|
+
return {
|
|
270
|
+
key: userAgent.key,
|
|
271
|
+
name: userAgent.name,
|
|
272
|
+
endpoint: userAgent.endpoint,
|
|
273
|
+
description: userAgent.description,
|
|
274
|
+
enabled: userAgent.enabled ?? true,
|
|
275
|
+
models: userAgent.models,
|
|
276
|
+
headers: userAgent.headers ?? {},
|
|
277
|
+
useTraces: userAgent.useTraces ?? false,
|
|
278
|
+
connectorType: userAgent.connectorType,
|
|
279
|
+
connectorConfig: userAgent.connectorConfig,
|
|
280
|
+
hooks: userAgent.hooks
|
|
281
|
+
};
|
|
282
|
+
}
|
|
283
|
+
function toModelConfig(userModel) {
|
|
284
|
+
return [
|
|
285
|
+
userModel.key,
|
|
286
|
+
{
|
|
287
|
+
model_id: userModel.model_id,
|
|
288
|
+
display_name: userModel.display_name,
|
|
289
|
+
provider: userModel.provider ?? "bedrock",
|
|
290
|
+
context_window: userModel.context_window ?? 2e5,
|
|
291
|
+
max_output_tokens: userModel.max_output_tokens ?? 4096
|
|
292
|
+
}
|
|
293
|
+
];
|
|
294
|
+
}
|
|
295
|
+
function mergeConfigs(userConfig, defaultConfig) {
|
|
296
|
+
const shouldExtend = userConfig.extends !== false;
|
|
297
|
+
let agents;
|
|
298
|
+
if (shouldExtend) {
|
|
299
|
+
const agentMap = /* @__PURE__ */ new Map();
|
|
300
|
+
for (const agent of defaultConfig.agents) {
|
|
301
|
+
agentMap.set(agent.key, agent);
|
|
302
|
+
}
|
|
303
|
+
for (const userAgent of userConfig.agents ?? []) {
|
|
304
|
+
agentMap.set(userAgent.key, toAgentConfig(userAgent));
|
|
305
|
+
}
|
|
306
|
+
agents = Array.from(agentMap.values());
|
|
307
|
+
} else {
|
|
308
|
+
agents = (userConfig.agents ?? []).map(toAgentConfig);
|
|
309
|
+
}
|
|
310
|
+
let models;
|
|
311
|
+
if (shouldExtend) {
|
|
312
|
+
models = { ...defaultConfig.models };
|
|
313
|
+
for (const userModel of userConfig.models ?? []) {
|
|
314
|
+
const [key, config] = toModelConfig(userModel);
|
|
315
|
+
models[key] = config;
|
|
316
|
+
}
|
|
317
|
+
} else {
|
|
318
|
+
models = {};
|
|
319
|
+
for (const userModel of userConfig.models ?? []) {
|
|
320
|
+
const [key, config] = toModelConfig(userModel);
|
|
321
|
+
models[key] = config;
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
const connectors = userConfig.connectors ?? [];
|
|
325
|
+
const testCases = userConfig.testCases ? Array.isArray(userConfig.testCases) ? userConfig.testCases : [userConfig.testCases] : [];
|
|
326
|
+
const reporters = userConfig.reporters ?? [["console"]];
|
|
327
|
+
const judge = userConfig.judge ?? {
|
|
328
|
+
provider: "bedrock",
|
|
329
|
+
model: "claude-sonnet-4"
|
|
330
|
+
};
|
|
331
|
+
const server = {
|
|
332
|
+
...DEFAULT_SERVER_CONFIG,
|
|
333
|
+
...userConfig.server
|
|
334
|
+
};
|
|
335
|
+
return {
|
|
336
|
+
server,
|
|
337
|
+
agents,
|
|
338
|
+
models,
|
|
339
|
+
connectors,
|
|
340
|
+
testCases,
|
|
341
|
+
reporters,
|
|
342
|
+
judge
|
|
343
|
+
};
|
|
344
|
+
}
|
|
345
|
+
async function loadUserConfig(configPath) {
|
|
346
|
+
try {
|
|
347
|
+
const fileUrl = pathToFileURL(configPath).href;
|
|
348
|
+
const module = await import(fileUrl);
|
|
349
|
+
return module.default ?? module;
|
|
350
|
+
} catch (error) {
|
|
351
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
352
|
+
throw new Error(`Failed to load config file ${configPath}: ${message}`);
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
var cachedConfig = null;
|
|
356
|
+
var cachedConfigPath = null;
|
|
357
|
+
async function loadConfig(cwd = process.cwd(), force = false) {
|
|
358
|
+
const configFile = findConfigFile(cwd);
|
|
359
|
+
if (!force && cachedConfig && cachedConfigPath === configFile?.path) {
|
|
360
|
+
return cachedConfig;
|
|
361
|
+
}
|
|
362
|
+
let userConfig = {};
|
|
363
|
+
if (configFile) {
|
|
364
|
+
console.log(`[Config] Loading ${configFile.path}`);
|
|
365
|
+
userConfig = await loadUserConfig(configFile.path);
|
|
366
|
+
} else {
|
|
367
|
+
console.log("[Config] No config file found, using defaults + environment variables");
|
|
368
|
+
}
|
|
369
|
+
const resolved = mergeConfigs(userConfig, DEFAULT_CONFIG);
|
|
370
|
+
cachedConfig = resolved;
|
|
371
|
+
cachedConfigPath = configFile?.path ?? null;
|
|
372
|
+
console.log(`[Config] Loaded ${resolved.agents.length} agents, ${Object.keys(resolved.models).length} models`);
|
|
373
|
+
return resolved;
|
|
374
|
+
}
|
|
375
|
+
function getConfigFileInfo(cwd = process.cwd()) {
|
|
376
|
+
return findConfigFile(cwd);
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
// services/connectors/registry.ts
|
|
380
|
+
var DEFAULT_CONNECTOR_TYPE = "agui-streaming";
|
|
381
|
+
var ConnectorRegistryImpl = class {
|
|
382
|
+
constructor() {
|
|
383
|
+
this.connectors = /* @__PURE__ */ new Map();
|
|
384
|
+
}
|
|
385
|
+
/**
|
|
386
|
+
* Register a connector implementation
|
|
387
|
+
* @throws Error if connector with same type is already registered
|
|
388
|
+
*/
|
|
389
|
+
register(connector) {
|
|
390
|
+
if (this.connectors.has(connector.type)) {
|
|
391
|
+
console.warn(
|
|
392
|
+
`[ConnectorRegistry] Overwriting existing connector for type: ${connector.type}`
|
|
393
|
+
);
|
|
394
|
+
}
|
|
395
|
+
this.connectors.set(connector.type, connector);
|
|
396
|
+
}
|
|
397
|
+
/**
|
|
398
|
+
* Get a connector by protocol type
|
|
399
|
+
*/
|
|
400
|
+
get(type) {
|
|
401
|
+
return this.connectors.get(type);
|
|
402
|
+
}
|
|
403
|
+
/**
|
|
404
|
+
* Get all registered connectors
|
|
405
|
+
*/
|
|
406
|
+
getAll() {
|
|
407
|
+
return Array.from(this.connectors.values());
|
|
408
|
+
}
|
|
409
|
+
/**
|
|
410
|
+
* Check if a connector is registered
|
|
411
|
+
*/
|
|
412
|
+
has(type) {
|
|
413
|
+
return this.connectors.has(type);
|
|
414
|
+
}
|
|
415
|
+
/**
|
|
416
|
+
* Get connector for an agent config
|
|
417
|
+
* Handles backwards compatibility with legacy configs
|
|
418
|
+
*
|
|
419
|
+
* Resolution order:
|
|
420
|
+
* 1. If endpoint starts with 'mock://', use mock connector
|
|
421
|
+
* 2. If connectorType is specified, use that
|
|
422
|
+
* 3. Default to 'agui-streaming'
|
|
423
|
+
*/
|
|
424
|
+
getForAgent(agent) {
|
|
425
|
+
if (agent.endpoint.startsWith("mock://")) {
|
|
426
|
+
const mockConnector2 = this.get("mock");
|
|
427
|
+
if (mockConnector2) {
|
|
428
|
+
return mockConnector2;
|
|
429
|
+
}
|
|
430
|
+
console.warn("[ConnectorRegistry] Mock connector not registered, falling back to default");
|
|
431
|
+
}
|
|
432
|
+
const connectorType = agent.connectorType ?? DEFAULT_CONNECTOR_TYPE;
|
|
433
|
+
const connector = this.get(connectorType);
|
|
434
|
+
if (!connector) {
|
|
435
|
+
console.error(
|
|
436
|
+
`[ConnectorRegistry] Connector not found for type: ${connectorType}, falling back to ${DEFAULT_CONNECTOR_TYPE}`
|
|
437
|
+
);
|
|
438
|
+
const defaultConnector = this.get(DEFAULT_CONNECTOR_TYPE);
|
|
439
|
+
if (!defaultConnector) {
|
|
440
|
+
throw new Error(
|
|
441
|
+
`No connector registered for type '${connectorType}' and no default connector available`
|
|
442
|
+
);
|
|
443
|
+
}
|
|
444
|
+
return defaultConnector;
|
|
445
|
+
}
|
|
446
|
+
return connector;
|
|
447
|
+
}
|
|
448
|
+
/**
|
|
449
|
+
* Clear all registered connectors (useful for testing)
|
|
450
|
+
*/
|
|
451
|
+
clear() {
|
|
452
|
+
this.connectors.clear();
|
|
453
|
+
}
|
|
454
|
+
/**
|
|
455
|
+
* Get list of registered connector types
|
|
456
|
+
*/
|
|
457
|
+
getRegisteredTypes() {
|
|
458
|
+
return Array.from(this.connectors.keys());
|
|
459
|
+
}
|
|
460
|
+
};
|
|
461
|
+
var connectorRegistry = new ConnectorRegistryImpl();
|
|
462
|
+
|
|
463
|
+
// lib/debug.ts
|
|
464
|
+
import fs from "fs";
|
|
465
|
+
import path from "path";
|
|
466
|
+
var isBrowser = typeof window !== "undefined";
|
|
467
|
+
var CONFIG_FILENAME = "agent-health.config.json";
|
|
468
|
+
var serverDebugEnabled = false;
|
|
469
|
+
if (!isBrowser) {
|
|
470
|
+
try {
|
|
471
|
+
const configPath = path.join(process.cwd(), CONFIG_FILENAME);
|
|
472
|
+
if (fs.existsSync(configPath)) {
|
|
473
|
+
const content = fs.readFileSync(configPath, "utf-8");
|
|
474
|
+
const config = JSON.parse(content) || {};
|
|
475
|
+
serverDebugEnabled = config.debug === true;
|
|
476
|
+
} else if (process.env?.DEBUG === "true") {
|
|
477
|
+
serverDebugEnabled = true;
|
|
478
|
+
}
|
|
479
|
+
} catch (err) {
|
|
480
|
+
if (process.env?.DEBUG === "true") {
|
|
481
|
+
serverDebugEnabled = true;
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
}
|
|
485
|
+
function isDebugEnabled() {
|
|
486
|
+
if (isBrowser) {
|
|
487
|
+
try {
|
|
488
|
+
return localStorage.getItem("agenteval_debug") === "true";
|
|
489
|
+
} catch {
|
|
490
|
+
return false;
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
return serverDebugEnabled;
|
|
494
|
+
}
|
|
495
|
+
function debug(module, ...args) {
|
|
496
|
+
if (isDebugEnabled()) {
|
|
497
|
+
console.debug(`[${module}]`, ...args);
|
|
498
|
+
}
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
// services/connectors/base/BaseConnector.ts
|
|
502
|
+
var BaseConnector = class {
|
|
503
|
+
/**
|
|
504
|
+
* Build HTTP headers from auth configuration
|
|
505
|
+
* @param auth Authentication configuration
|
|
506
|
+
* @returns Headers object ready for fetch/axios
|
|
507
|
+
*/
|
|
508
|
+
buildAuthHeaders(auth) {
|
|
509
|
+
const headers = {};
|
|
510
|
+
switch (auth.type) {
|
|
511
|
+
case "basic":
|
|
512
|
+
if (auth.username && auth.password) {
|
|
513
|
+
const credentials = Buffer.from(`${auth.username}:${auth.password}`).toString("base64");
|
|
514
|
+
headers["Authorization"] = `Basic ${credentials}`;
|
|
515
|
+
}
|
|
516
|
+
break;
|
|
517
|
+
case "bearer":
|
|
518
|
+
if (auth.token) {
|
|
519
|
+
headers["Authorization"] = `Bearer ${auth.token}`;
|
|
520
|
+
}
|
|
521
|
+
break;
|
|
522
|
+
case "api-key":
|
|
523
|
+
if (auth.token) {
|
|
524
|
+
headers["X-API-Key"] = auth.token;
|
|
525
|
+
headers["x-api-key"] = auth.token;
|
|
526
|
+
}
|
|
527
|
+
break;
|
|
528
|
+
case "aws-sigv4":
|
|
529
|
+
console.warn("[BaseConnector] AWS SigV4 auth requires runtime signing");
|
|
530
|
+
break;
|
|
531
|
+
case "none":
|
|
532
|
+
default:
|
|
533
|
+
break;
|
|
534
|
+
}
|
|
535
|
+
if (auth.headers) {
|
|
536
|
+
Object.assign(headers, auth.headers);
|
|
537
|
+
}
|
|
538
|
+
return headers;
|
|
539
|
+
}
|
|
540
|
+
/**
|
|
541
|
+
* Build environment variables from auth configuration
|
|
542
|
+
* Used by subprocess connectors
|
|
543
|
+
*/
|
|
544
|
+
buildAuthEnv(auth) {
|
|
545
|
+
const env = {};
|
|
546
|
+
if (auth.type === "aws-sigv4") {
|
|
547
|
+
if (auth.awsRegion) env["AWS_REGION"] = auth.awsRegion;
|
|
548
|
+
if (auth.awsAccessKeyId) env["AWS_ACCESS_KEY_ID"] = auth.awsAccessKeyId;
|
|
549
|
+
if (auth.awsSecretAccessKey) env["AWS_SECRET_ACCESS_KEY"] = auth.awsSecretAccessKey;
|
|
550
|
+
if (auth.awsSessionToken) env["AWS_SESSION_TOKEN"] = auth.awsSessionToken;
|
|
551
|
+
}
|
|
552
|
+
return env;
|
|
553
|
+
}
|
|
554
|
+
/**
|
|
555
|
+
* Generate a unique ID for trajectory steps
|
|
556
|
+
*/
|
|
557
|
+
generateId() {
|
|
558
|
+
return `${Date.now()}-${Math.random().toString(36).substring(2, 9)}`;
|
|
559
|
+
}
|
|
560
|
+
/**
|
|
561
|
+
* Create a trajectory step with common fields
|
|
562
|
+
*/
|
|
563
|
+
createStep(type, content, extra) {
|
|
564
|
+
return {
|
|
565
|
+
id: this.generateId(),
|
|
566
|
+
timestamp: Date.now(),
|
|
567
|
+
type,
|
|
568
|
+
content,
|
|
569
|
+
...extra
|
|
570
|
+
};
|
|
571
|
+
}
|
|
572
|
+
/**
|
|
573
|
+
* Default health check implementation
|
|
574
|
+
* Subclasses can override for protocol-specific checks
|
|
575
|
+
*/
|
|
576
|
+
async healthCheck(endpoint, auth) {
|
|
577
|
+
try {
|
|
578
|
+
const headers = this.buildAuthHeaders(auth);
|
|
579
|
+
const response = await fetch(endpoint, {
|
|
580
|
+
method: "HEAD",
|
|
581
|
+
headers
|
|
582
|
+
});
|
|
583
|
+
return response.ok;
|
|
584
|
+
} catch (error) {
|
|
585
|
+
console.error(`[${this.type}] Health check failed:`, error);
|
|
586
|
+
return false;
|
|
587
|
+
}
|
|
588
|
+
}
|
|
589
|
+
/**
|
|
590
|
+
* Log debug message with connector type prefix
|
|
591
|
+
*/
|
|
592
|
+
debug(message, ...args) {
|
|
593
|
+
debug(this.type, message, ...args);
|
|
594
|
+
}
|
|
595
|
+
/**
|
|
596
|
+
* Log error message with connector type prefix
|
|
597
|
+
*/
|
|
598
|
+
error(message, ...args) {
|
|
599
|
+
console.error(`[${this.type}] ${message}`, ...args);
|
|
600
|
+
}
|
|
601
|
+
};
|
|
602
|
+
|
|
603
|
+
// types/agui.ts
|
|
604
|
+
import { EventType } from "@ag-ui/core";
|
|
605
|
+
var AGUIEventType = EventType;
|
|
606
|
+
|
|
607
|
+
// services/agent/sseStream.ts
|
|
608
|
+
var SSEClient = class {
|
|
609
|
+
constructor() {
|
|
610
|
+
this.abortController = null;
|
|
611
|
+
}
|
|
612
|
+
/**
|
|
613
|
+
* Start consuming SSE stream from the agent endpoint
|
|
614
|
+
*/
|
|
615
|
+
async consume(options) {
|
|
616
|
+
const {
|
|
617
|
+
url,
|
|
618
|
+
method = "POST",
|
|
619
|
+
headers = {},
|
|
620
|
+
body,
|
|
621
|
+
onEvent,
|
|
622
|
+
onError,
|
|
623
|
+
onComplete,
|
|
624
|
+
completeOnRunEnd = false,
|
|
625
|
+
idleTimeoutMs = 12e4
|
|
626
|
+
// 2 minute idle timeout by default (LLM agents can be slow)
|
|
627
|
+
} = options;
|
|
628
|
+
this.abortController = new AbortController();
|
|
629
|
+
debug("SSE", "Connecting to", url);
|
|
630
|
+
debug("SSE", "Method:", method);
|
|
631
|
+
debug("SSE", "Headers:", headers);
|
|
632
|
+
debug("SSE", "Payload:", body ? JSON.stringify(body, null, 2).substring(0, 500) : "none");
|
|
633
|
+
debug("SSE", "Timeout:", idleTimeoutMs, "ms");
|
|
634
|
+
try {
|
|
635
|
+
const requestConfig = {
|
|
636
|
+
method,
|
|
637
|
+
headers: {
|
|
638
|
+
"Content-Type": "application/json",
|
|
639
|
+
Accept: "text/event-stream",
|
|
640
|
+
...headers
|
|
641
|
+
},
|
|
642
|
+
body: body ? JSON.stringify(body) : void 0,
|
|
643
|
+
signal: this.abortController.signal
|
|
644
|
+
};
|
|
645
|
+
debug("SSE", "Request config:", JSON.stringify(requestConfig, null, 2).substring(0, 500));
|
|
646
|
+
const response = await fetch(url, requestConfig);
|
|
647
|
+
debug("SSE", "Response received:", response.status, response.statusText);
|
|
648
|
+
if (!response.ok) {
|
|
649
|
+
let errorBody = "";
|
|
650
|
+
try {
|
|
651
|
+
errorBody = await response.text();
|
|
652
|
+
debug("SSE", "Error response body:", errorBody.substring(0, 500));
|
|
653
|
+
} catch {
|
|
654
|
+
debug("SSE", "Could not read error response body");
|
|
655
|
+
}
|
|
656
|
+
throw new Error(`HTTP ${response.status}: ${response.statusText}${errorBody ? ` - ${errorBody}` : ""}`);
|
|
657
|
+
}
|
|
658
|
+
if (!response.body) {
|
|
659
|
+
throw new Error("Response body is null");
|
|
660
|
+
}
|
|
661
|
+
console.info("[SSE] Connected to agent endpoint, streaming events...");
|
|
662
|
+
debug("SSE", "Response status:", response.status);
|
|
663
|
+
debug("SSE", "Content-Type:", response.headers.get("content-type"));
|
|
664
|
+
const completionReason = await this.processStream(response.body, onEvent, completeOnRunEnd, idleTimeoutMs);
|
|
665
|
+
console.info(`[SSE] Stream completed: ${completionReason}`);
|
|
666
|
+
debug("SSE", `Stream completed: ${completionReason}`);
|
|
667
|
+
onComplete?.();
|
|
668
|
+
} catch (error) {
|
|
669
|
+
if (error instanceof Error) {
|
|
670
|
+
if (error.name === "AbortError") {
|
|
671
|
+
debug("SSE", "Stream aborted (expected after run completion)");
|
|
672
|
+
onComplete?.();
|
|
673
|
+
} else {
|
|
674
|
+
console.error("[SSE] Stream error:", error.message);
|
|
675
|
+
console.error("[SSE] Debug mode is:", isDebugEnabled() ? "ENABLED \u2705" : "DISABLED \u274C");
|
|
676
|
+
const errorDetails = {
|
|
677
|
+
name: error.name,
|
|
678
|
+
message: error.message,
|
|
679
|
+
stack: error.stack,
|
|
680
|
+
cause: error.cause,
|
|
681
|
+
url,
|
|
682
|
+
method,
|
|
683
|
+
headers
|
|
684
|
+
};
|
|
685
|
+
debug("SSE", "Error details:", errorDetails);
|
|
686
|
+
if (error.message.includes("fetch failed")) {
|
|
687
|
+
debug("SSE", '\u{1F4A1} Diagnostic: "fetch failed" typically means:');
|
|
688
|
+
debug("SSE", " - Connection refused (endpoint not running)");
|
|
689
|
+
debug("SSE", " - DNS resolution failed (invalid hostname)");
|
|
690
|
+
debug("SSE", " - Network unreachable (firewall/VPN issues)");
|
|
691
|
+
debug("SSE", " - SSL/TLS certificate issues (self-signed cert)");
|
|
692
|
+
debug("SSE", ` Check if ${url} is accessible`);
|
|
693
|
+
} else if (error.message.includes("timeout")) {
|
|
694
|
+
debug("SSE", "\u{1F4A1} Diagnostic: Request timed out - endpoint may be slow or unresponsive");
|
|
695
|
+
} else if (error.message.includes("ENOTFOUND")) {
|
|
696
|
+
debug("SSE", "\u{1F4A1} Diagnostic: DNS lookup failed - hostname not found");
|
|
697
|
+
} else if (error.message.includes("ECONNREFUSED")) {
|
|
698
|
+
debug("SSE", "\u{1F4A1} Diagnostic: Connection refused - service not listening on this port");
|
|
699
|
+
}
|
|
700
|
+
onError?.(error);
|
|
701
|
+
}
|
|
702
|
+
} else {
|
|
703
|
+
console.error("[SSE] Unknown error:", error);
|
|
704
|
+
debug("SSE", "Unknown error details:", error);
|
|
705
|
+
onError?.(new Error("Unknown error occurred"));
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
}
|
|
709
|
+
/**
|
|
710
|
+
* Process the ReadableStream and parse SSE events
|
|
711
|
+
* @returns Reason for stream completion
|
|
712
|
+
*/
|
|
713
|
+
async processStream(stream, onEvent, completeOnRunEnd = false, idleTimeoutMs = 12e4) {
|
|
714
|
+
const reader = stream.getReader();
|
|
715
|
+
const decoder = new TextDecoder();
|
|
716
|
+
let buffer = "";
|
|
717
|
+
let lastEventTime = Date.now();
|
|
718
|
+
let eventCount = 0;
|
|
719
|
+
let idleCheckInterval = null;
|
|
720
|
+
const idleTimeoutPromise = new Promise((resolve5) => {
|
|
721
|
+
idleCheckInterval = setInterval(() => {
|
|
722
|
+
const idleTime = Date.now() - lastEventTime;
|
|
723
|
+
if (eventCount > 0 && idleTime > idleTimeoutMs) {
|
|
724
|
+
debug("SSE", `Idle timeout: no events for ${idleTime}ms (threshold: ${idleTimeoutMs}ms)`);
|
|
725
|
+
this.abort();
|
|
726
|
+
resolve5("idle_timeout");
|
|
727
|
+
}
|
|
728
|
+
}, 1e3);
|
|
729
|
+
});
|
|
730
|
+
try {
|
|
731
|
+
const streamPromise = (async () => {
|
|
732
|
+
while (true) {
|
|
733
|
+
const { done, value } = await reader.read();
|
|
734
|
+
if (done) {
|
|
735
|
+
return "connection_closed";
|
|
736
|
+
}
|
|
737
|
+
lastEventTime = Date.now();
|
|
738
|
+
buffer += decoder.decode(value, { stream: true });
|
|
739
|
+
const lines = buffer.split("\n");
|
|
740
|
+
buffer = lines.pop() || "";
|
|
741
|
+
for (const line of lines) {
|
|
742
|
+
if (line.startsWith("data: ")) {
|
|
743
|
+
const data = line.slice(6);
|
|
744
|
+
if (data.trim()) {
|
|
745
|
+
debug("SSE", "Raw event:", data.substring(0, 200) + (data.length > 200 ? "..." : ""));
|
|
746
|
+
try {
|
|
747
|
+
const event = JSON.parse(data);
|
|
748
|
+
debug("SSE", "Parsed event:", event.type);
|
|
749
|
+
eventCount++;
|
|
750
|
+
if (eventCount % 10 === 0) {
|
|
751
|
+
console.info(`[SSE] Processed ${eventCount} events...`);
|
|
752
|
+
}
|
|
753
|
+
onEvent(event);
|
|
754
|
+
if (completeOnRunEnd && (event.type === AGUIEventType.RUN_FINISHED || event.type === AGUIEventType.RUN_ERROR)) {
|
|
755
|
+
debug("SSE", `Received ${event.type}, completing stream`);
|
|
756
|
+
this.abort();
|
|
757
|
+
return `event:${event.type}`;
|
|
758
|
+
}
|
|
759
|
+
} catch (parseError) {
|
|
760
|
+
console.error("[SSE] Parse error:", parseError);
|
|
761
|
+
debug("SSE", "Failed data:", data);
|
|
762
|
+
}
|
|
763
|
+
}
|
|
764
|
+
}
|
|
765
|
+
}
|
|
766
|
+
}
|
|
767
|
+
})();
|
|
768
|
+
const reason = await Promise.race([streamPromise, idleTimeoutPromise]);
|
|
769
|
+
return reason;
|
|
770
|
+
} finally {
|
|
771
|
+
if (idleCheckInterval) {
|
|
772
|
+
clearInterval(idleCheckInterval);
|
|
773
|
+
}
|
|
774
|
+
reader.releaseLock();
|
|
775
|
+
}
|
|
776
|
+
}
|
|
777
|
+
/**
|
|
778
|
+
* Abort the current stream connection
|
|
779
|
+
*/
|
|
780
|
+
abort() {
|
|
781
|
+
this.abortController?.abort();
|
|
782
|
+
}
|
|
783
|
+
};
|
|
784
|
+
async function consumeSSEStream(url, payload, onEvent, headers, options) {
|
|
785
|
+
const client = new SSEClient();
|
|
786
|
+
return new Promise((resolve5, reject) => {
|
|
787
|
+
client.consume({
|
|
788
|
+
url,
|
|
789
|
+
method: "POST",
|
|
790
|
+
headers,
|
|
791
|
+
body: payload,
|
|
792
|
+
onEvent,
|
|
793
|
+
onError: (error) => reject(error),
|
|
794
|
+
onComplete: () => resolve5(),
|
|
795
|
+
// Enable auto-completion on RUN_FINISHED/RUN_ERROR events
|
|
796
|
+
// This prevents hanging when the agent doesn't close the connection
|
|
797
|
+
completeOnRunEnd: true,
|
|
798
|
+
idleTimeoutMs: options?.idleTimeoutMs
|
|
799
|
+
});
|
|
800
|
+
});
|
|
801
|
+
}
|
|
802
|
+
|
|
803
|
+
// services/agent/payloadBuilder.ts
|
|
804
|
+
var DEFAULT_PPL_TOOL = {
|
|
805
|
+
name: "execute_ppl_query",
|
|
806
|
+
description: "Update the query bar with a PPL query and optionally execute it",
|
|
807
|
+
parameters: {
|
|
808
|
+
type: "object",
|
|
809
|
+
properties: {
|
|
810
|
+
query: {
|
|
811
|
+
type: "string",
|
|
812
|
+
description: "The PPL query to set in the query bar"
|
|
813
|
+
},
|
|
814
|
+
autoExecute: {
|
|
815
|
+
type: "boolean",
|
|
816
|
+
description: "Whether to automatically execute the query (default: true)"
|
|
817
|
+
},
|
|
818
|
+
description: {
|
|
819
|
+
type: "string",
|
|
820
|
+
description: "Optional description of what the query does"
|
|
821
|
+
}
|
|
822
|
+
},
|
|
823
|
+
required: ["query"]
|
|
824
|
+
}
|
|
825
|
+
};
|
|
826
|
+
function generateId(prefix) {
|
|
827
|
+
const timestamp = Date.now();
|
|
828
|
+
const random = Math.random().toString(36).substring(2, 11);
|
|
829
|
+
return `${prefix}-${timestamp}-${random}`;
|
|
830
|
+
}
|
|
831
|
+
function buildAgentPayload(testCase, modelId, threadId, runId) {
|
|
832
|
+
const tools = testCase.tools || [DEFAULT_PPL_TOOL];
|
|
833
|
+
return {
|
|
834
|
+
threadId: threadId || generateId("thread"),
|
|
835
|
+
runId: runId || generateId("run"),
|
|
836
|
+
messages: [
|
|
837
|
+
{
|
|
838
|
+
id: generateId("msg"),
|
|
839
|
+
role: "user",
|
|
840
|
+
content: testCase.initialPrompt
|
|
841
|
+
}
|
|
842
|
+
],
|
|
843
|
+
tools,
|
|
844
|
+
context: testCase.context || [],
|
|
845
|
+
state: {},
|
|
846
|
+
forwardedProps: {}
|
|
847
|
+
};
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
// services/agent/aguiConverter.ts
|
|
851
|
+
import { v4 as uuidv4 } from "uuid";
|
|
852
|
+
var AGUIToTrajectoryConverter = class {
|
|
853
|
+
constructor() {
|
|
854
|
+
this.currentTextMessage = null;
|
|
855
|
+
this.activeTools = /* @__PURE__ */ new Map();
|
|
856
|
+
this.hasEmittedAction = false;
|
|
857
|
+
this.runFinished = false;
|
|
858
|
+
this.pendingTextIsResponse = false;
|
|
859
|
+
this.runId = null;
|
|
860
|
+
this.threadId = null;
|
|
861
|
+
// Thinking state tracking
|
|
862
|
+
this.isThinking = false;
|
|
863
|
+
this.currentThinkingMessage = null;
|
|
864
|
+
}
|
|
865
|
+
/**
|
|
866
|
+
* Process a single AG UI event and convert it to TrajectoryStep(s)
|
|
867
|
+
* Returns an array of steps (usually 0 or 1, sometimes more)
|
|
868
|
+
*/
|
|
869
|
+
processEvent(event) {
|
|
870
|
+
debug("Converter", `Event: ${event.type}`, JSON.stringify(event).substring(0, 300));
|
|
871
|
+
let steps = [];
|
|
872
|
+
switch (event.type) {
|
|
873
|
+
case AGUIEventType.RUN_STARTED:
|
|
874
|
+
steps = this.handleRunStarted(event);
|
|
875
|
+
break;
|
|
876
|
+
case AGUIEventType.RUN_FINISHED:
|
|
877
|
+
steps = this.handleRunFinished(event);
|
|
878
|
+
break;
|
|
879
|
+
case AGUIEventType.RUN_ERROR:
|
|
880
|
+
steps = this.handleRunError(event);
|
|
881
|
+
break;
|
|
882
|
+
case AGUIEventType.TEXT_MESSAGE_START:
|
|
883
|
+
steps = this.handleTextMessageStart(event);
|
|
884
|
+
break;
|
|
885
|
+
case AGUIEventType.TEXT_MESSAGE_CONTENT:
|
|
886
|
+
steps = this.handleTextMessageContent(event);
|
|
887
|
+
break;
|
|
888
|
+
case AGUIEventType.TEXT_MESSAGE_END:
|
|
889
|
+
steps = this.handleTextMessageEnd(event);
|
|
890
|
+
break;
|
|
891
|
+
case AGUIEventType.ACTIVITY_SNAPSHOT:
|
|
892
|
+
steps = this.handleActivitySnapshot(event);
|
|
893
|
+
break;
|
|
894
|
+
case AGUIEventType.ACTIVITY_DELTA:
|
|
895
|
+
steps = this.handleActivityDelta(event);
|
|
896
|
+
break;
|
|
897
|
+
case AGUIEventType.TOOL_CALL_START:
|
|
898
|
+
steps = this.handleToolCallStart(event);
|
|
899
|
+
break;
|
|
900
|
+
case AGUIEventType.TOOL_CALL_ARGS:
|
|
901
|
+
steps = this.handleToolCallArgs(event);
|
|
902
|
+
break;
|
|
903
|
+
case AGUIEventType.TOOL_CALL_END:
|
|
904
|
+
steps = this.handleToolCallEnd(event);
|
|
905
|
+
break;
|
|
906
|
+
case AGUIEventType.TOOL_CALL_RESULT:
|
|
907
|
+
steps = this.handleToolCallResult(event);
|
|
908
|
+
break;
|
|
909
|
+
// Thinking events - extended reasoning from the model
|
|
910
|
+
case AGUIEventType.THINKING_START:
|
|
911
|
+
steps = this.handleThinkingStart(event);
|
|
912
|
+
break;
|
|
913
|
+
case AGUIEventType.THINKING_END:
|
|
914
|
+
steps = this.handleThinkingEnd(event);
|
|
915
|
+
break;
|
|
916
|
+
case AGUIEventType.THINKING_TEXT_MESSAGE_START:
|
|
917
|
+
steps = this.handleThinkingTextMessageStart(event);
|
|
918
|
+
break;
|
|
919
|
+
case AGUIEventType.THINKING_TEXT_MESSAGE_CONTENT:
|
|
920
|
+
steps = this.handleThinkingTextMessageContent(event);
|
|
921
|
+
break;
|
|
922
|
+
case AGUIEventType.THINKING_TEXT_MESSAGE_END:
|
|
923
|
+
steps = this.handleThinkingTextMessageEnd(event);
|
|
924
|
+
break;
|
|
925
|
+
default:
|
|
926
|
+
debug("Converter", `Skipped unhandled event: ${event.type}`);
|
|
927
|
+
break;
|
|
928
|
+
}
|
|
929
|
+
if (steps.length > 0) {
|
|
930
|
+
debug("Converter", `Generated ${steps.length} step(s):`, steps.map((s) => `${s.type}${s.toolName ? `:${s.toolName}` : ""}`).join(", "));
|
|
931
|
+
}
|
|
932
|
+
return steps;
|
|
933
|
+
}
|
|
934
|
+
handleRunStarted(event) {
|
|
935
|
+
this.runId = event.runId;
|
|
936
|
+
this.threadId = event.threadId;
|
|
937
|
+
debug("Converter", `Run started - runId: ${this.runId}, threadId: ${this.threadId}`);
|
|
938
|
+
this.currentTextMessage = null;
|
|
939
|
+
this.activeTools.clear();
|
|
940
|
+
this.hasEmittedAction = false;
|
|
941
|
+
this.runFinished = false;
|
|
942
|
+
this.pendingTextIsResponse = false;
|
|
943
|
+
this.isThinking = false;
|
|
944
|
+
this.currentThinkingMessage = null;
|
|
945
|
+
return [];
|
|
946
|
+
}
|
|
947
|
+
getRunId() {
|
|
948
|
+
return this.runId;
|
|
949
|
+
}
|
|
950
|
+
getThreadId() {
|
|
951
|
+
return this.threadId;
|
|
952
|
+
}
|
|
953
|
+
handleRunFinished(event) {
|
|
954
|
+
this.runFinished = true;
|
|
955
|
+
if (this.currentTextMessage) {
|
|
956
|
+
this.pendingTextIsResponse = true;
|
|
957
|
+
}
|
|
958
|
+
return [];
|
|
959
|
+
}
|
|
960
|
+
handleRunError(event) {
|
|
961
|
+
console.error("[Converter] Run error:", event.message);
|
|
962
|
+
return [{
|
|
963
|
+
id: uuidv4(),
|
|
964
|
+
timestamp: event.timestamp,
|
|
965
|
+
type: "tool_result",
|
|
966
|
+
content: `Error: ${event.message}`,
|
|
967
|
+
status: "FAILURE" /* FAILURE */
|
|
968
|
+
}];
|
|
969
|
+
}
|
|
970
|
+
handleTextMessageStart(event) {
|
|
971
|
+
debug("Converter", `Text message start: ${event.messageId}`);
|
|
972
|
+
this.currentTextMessage = {
|
|
973
|
+
messageId: event.messageId,
|
|
974
|
+
startTime: event.timestamp,
|
|
975
|
+
content: ""
|
|
976
|
+
};
|
|
977
|
+
return [];
|
|
978
|
+
}
|
|
979
|
+
handleTextMessageContent(event) {
|
|
980
|
+
if (this.currentTextMessage && this.currentTextMessage.messageId === event.messageId) {
|
|
981
|
+
this.currentTextMessage.content += event.delta;
|
|
982
|
+
}
|
|
983
|
+
return [];
|
|
984
|
+
}
|
|
985
|
+
handleTextMessageEnd(event) {
|
|
986
|
+
if (!this.currentTextMessage || this.currentTextMessage.messageId !== event.messageId) {
|
|
987
|
+
debug("Converter", `Text end for unknown message: ${event.messageId}`);
|
|
988
|
+
return [];
|
|
989
|
+
}
|
|
990
|
+
const latencyMs = event.timestamp - this.currentTextMessage.startTime;
|
|
991
|
+
const content = this.currentTextMessage.content.trim();
|
|
992
|
+
let stepType;
|
|
993
|
+
if (this.pendingTextIsResponse || this.runFinished) {
|
|
994
|
+
stepType = "response";
|
|
995
|
+
} else {
|
|
996
|
+
stepType = "assistant";
|
|
997
|
+
}
|
|
998
|
+
debug("Converter", `Classification: ${stepType} (runFinished=${this.runFinished}, pendingResponse=${this.pendingTextIsResponse}, hasAction=${this.hasEmittedAction})`);
|
|
999
|
+
if (stepType === "assistant" && content.length === 0) {
|
|
1000
|
+
debug("Converter", "Skipping empty assistant message");
|
|
1001
|
+
this.currentTextMessage = null;
|
|
1002
|
+
return [];
|
|
1003
|
+
}
|
|
1004
|
+
const step = {
|
|
1005
|
+
id: uuidv4(),
|
|
1006
|
+
timestamp: event.timestamp,
|
|
1007
|
+
type: stepType,
|
|
1008
|
+
content,
|
|
1009
|
+
latencyMs
|
|
1010
|
+
};
|
|
1011
|
+
this.currentTextMessage = null;
|
|
1012
|
+
return [step];
|
|
1013
|
+
}
|
|
1014
|
+
handleActivitySnapshot(event) {
|
|
1015
|
+
const toolName = this.extractToolName(event.content.title);
|
|
1016
|
+
const toolArgs = this.parseToolArgs(event.content.description);
|
|
1017
|
+
const actionStepId = uuidv4();
|
|
1018
|
+
this.activeTools.set(event.messageId, {
|
|
1019
|
+
messageId: event.messageId,
|
|
1020
|
+
toolName,
|
|
1021
|
+
toolArgs,
|
|
1022
|
+
argsAccumulator: "",
|
|
1023
|
+
startTime: event.timestamp,
|
|
1024
|
+
actionStepId
|
|
1025
|
+
});
|
|
1026
|
+
this.hasEmittedAction = true;
|
|
1027
|
+
debug("Converter", `Tool action: ${toolName}`, toolArgs);
|
|
1028
|
+
return [{
|
|
1029
|
+
id: actionStepId,
|
|
1030
|
+
timestamp: event.timestamp,
|
|
1031
|
+
type: "action",
|
|
1032
|
+
content: `Calling ${toolName}...`,
|
|
1033
|
+
toolName,
|
|
1034
|
+
toolArgs
|
|
1035
|
+
}];
|
|
1036
|
+
}
|
|
1037
|
+
handleActivityDelta(event) {
|
|
1038
|
+
const toolState = this.activeTools.get(event.messageId);
|
|
1039
|
+
if (!toolState) return [];
|
|
1040
|
+
const isCompletion = event.patch.some(
|
|
1041
|
+
(op) => op.path === "/icon" && (op.value === "CheckCircle" || op.value === "Check")
|
|
1042
|
+
);
|
|
1043
|
+
if (!isCompletion) return [];
|
|
1044
|
+
const descriptionPatch = event.patch.find((op) => op.path === "/description");
|
|
1045
|
+
const resultContent = descriptionPatch?.value || "Tool execution completed";
|
|
1046
|
+
const latencyMs = event.timestamp - toolState.startTime;
|
|
1047
|
+
this.activeTools.delete(event.messageId);
|
|
1048
|
+
return [{
|
|
1049
|
+
id: uuidv4(),
|
|
1050
|
+
timestamp: event.timestamp,
|
|
1051
|
+
type: "tool_result",
|
|
1052
|
+
content: resultContent,
|
|
1053
|
+
status: "SUCCESS" /* SUCCESS */,
|
|
1054
|
+
latencyMs
|
|
1055
|
+
}];
|
|
1056
|
+
}
|
|
1057
|
+
handleToolCallStart(event) {
|
|
1058
|
+
const actionStepId = uuidv4();
|
|
1059
|
+
this.activeTools.set(event.toolCallId, {
|
|
1060
|
+
messageId: event.toolCallId,
|
|
1061
|
+
toolName: event.toolCallName,
|
|
1062
|
+
toolArgs: {},
|
|
1063
|
+
argsAccumulator: "",
|
|
1064
|
+
// Will accumulate delta strings
|
|
1065
|
+
startTime: event.timestamp,
|
|
1066
|
+
actionStepId
|
|
1067
|
+
});
|
|
1068
|
+
this.hasEmittedAction = true;
|
|
1069
|
+
debug("Converter", `Tool call started: ${event.toolCallName} (${event.toolCallId})`);
|
|
1070
|
+
return [];
|
|
1071
|
+
}
|
|
1072
|
+
handleToolCallArgs(event) {
|
|
1073
|
+
const toolState = this.activeTools.get(event.toolCallId);
|
|
1074
|
+
if (toolState) {
|
|
1075
|
+
toolState.argsAccumulator += event.delta;
|
|
1076
|
+
debug("Converter", `Tool args delta accumulated (${toolState.argsAccumulator.length} chars total)`);
|
|
1077
|
+
}
|
|
1078
|
+
return [];
|
|
1079
|
+
}
|
|
1080
|
+
/**
|
|
1081
|
+
* Handle TOOL_CALL_END - emits the action step with complete args
|
|
1082
|
+
* This is called when the agent is done sending tool call arguments
|
|
1083
|
+
* and expects the client to execute the tool
|
|
1084
|
+
*/
|
|
1085
|
+
handleToolCallEnd(event) {
|
|
1086
|
+
const toolState = this.activeTools.get(event.toolCallId);
|
|
1087
|
+
if (!toolState) {
|
|
1088
|
+
debug("Converter", `Tool call end for unknown tool: ${event.toolCallId}`);
|
|
1089
|
+
return [];
|
|
1090
|
+
}
|
|
1091
|
+
let parsedArgs = {};
|
|
1092
|
+
if (toolState.argsAccumulator) {
|
|
1093
|
+
try {
|
|
1094
|
+
parsedArgs = JSON.parse(toolState.argsAccumulator);
|
|
1095
|
+
debug("Converter", `Tool args parsed: ${JSON.stringify(parsedArgs).substring(0, 200)}`);
|
|
1096
|
+
} catch {
|
|
1097
|
+
parsedArgs = { _raw: toolState.argsAccumulator };
|
|
1098
|
+
}
|
|
1099
|
+
}
|
|
1100
|
+
toolState.toolArgs = parsedArgs;
|
|
1101
|
+
const latencyMs = event.timestamp - toolState.startTime;
|
|
1102
|
+
const actionStep = {
|
|
1103
|
+
id: toolState.actionStepId,
|
|
1104
|
+
timestamp: toolState.startTime,
|
|
1105
|
+
type: "action",
|
|
1106
|
+
content: `Calling ${toolState.toolName}...`,
|
|
1107
|
+
toolName: toolState.toolName,
|
|
1108
|
+
toolArgs: parsedArgs,
|
|
1109
|
+
latencyMs
|
|
1110
|
+
};
|
|
1111
|
+
debug("Converter", `Tool call end: ${toolState.toolName} - action step emitted with args`);
|
|
1112
|
+
toolState.actionStepId = "";
|
|
1113
|
+
return [actionStep];
|
|
1114
|
+
}
|
|
1115
|
+
handleToolCallResult(event) {
|
|
1116
|
+
const toolState = this.activeTools.get(event.toolCallId);
|
|
1117
|
+
if (!toolState) return [];
|
|
1118
|
+
const latencyMs = event.timestamp - toolState.startTime;
|
|
1119
|
+
const steps = [];
|
|
1120
|
+
const actionAlreadyEmitted = !toolState.actionStepId;
|
|
1121
|
+
if (!actionAlreadyEmitted) {
|
|
1122
|
+
let parsedArgs = {};
|
|
1123
|
+
if (toolState.argsAccumulator) {
|
|
1124
|
+
try {
|
|
1125
|
+
parsedArgs = JSON.parse(toolState.argsAccumulator);
|
|
1126
|
+
debug("Converter", `Tool args parsed: ${JSON.stringify(parsedArgs).substring(0, 200)}`);
|
|
1127
|
+
} catch {
|
|
1128
|
+
parsedArgs = { _raw: toolState.argsAccumulator };
|
|
1129
|
+
}
|
|
1130
|
+
}
|
|
1131
|
+
toolState.toolArgs = parsedArgs;
|
|
1132
|
+
steps.push({
|
|
1133
|
+
id: toolState.actionStepId,
|
|
1134
|
+
timestamp: toolState.startTime,
|
|
1135
|
+
type: "action",
|
|
1136
|
+
content: `Calling ${toolState.toolName}...`,
|
|
1137
|
+
toolName: toolState.toolName,
|
|
1138
|
+
toolArgs: parsedArgs
|
|
1139
|
+
});
|
|
1140
|
+
}
|
|
1141
|
+
let resultContent;
|
|
1142
|
+
try {
|
|
1143
|
+
const parsed = JSON.parse(event.content);
|
|
1144
|
+
resultContent = typeof parsed === "string" ? parsed : JSON.stringify(parsed, null, 2);
|
|
1145
|
+
} catch (e) {
|
|
1146
|
+
resultContent = event.content;
|
|
1147
|
+
}
|
|
1148
|
+
steps.push({
|
|
1149
|
+
id: uuidv4(),
|
|
1150
|
+
timestamp: event.timestamp,
|
|
1151
|
+
type: "tool_result",
|
|
1152
|
+
content: resultContent,
|
|
1153
|
+
status: "SUCCESS" /* SUCCESS */,
|
|
1154
|
+
latencyMs
|
|
1155
|
+
});
|
|
1156
|
+
this.activeTools.delete(event.toolCallId);
|
|
1157
|
+
debug("Converter", `Tool call result: ${toolState.toolName} -> ${steps.length} steps emitted (action already emitted: ${actionAlreadyEmitted})`);
|
|
1158
|
+
return steps;
|
|
1159
|
+
}
|
|
1160
|
+
// ============ THINKING Event Handlers ============
|
|
1161
|
+
handleThinkingStart(event) {
|
|
1162
|
+
this.isThinking = true;
|
|
1163
|
+
debug("Converter", "Thinking started");
|
|
1164
|
+
return [];
|
|
1165
|
+
}
|
|
1166
|
+
handleThinkingEnd(event) {
|
|
1167
|
+
this.isThinking = false;
|
|
1168
|
+
debug("Converter", "Thinking ended");
|
|
1169
|
+
return [];
|
|
1170
|
+
}
|
|
1171
|
+
handleThinkingTextMessageStart(event) {
|
|
1172
|
+
this.currentThinkingMessage = {
|
|
1173
|
+
startTime: event.timestamp || Date.now(),
|
|
1174
|
+
content: ""
|
|
1175
|
+
};
|
|
1176
|
+
debug("Converter", "Thinking text message started");
|
|
1177
|
+
return [];
|
|
1178
|
+
}
|
|
1179
|
+
handleThinkingTextMessageContent(event) {
|
|
1180
|
+
if (this.currentThinkingMessage) {
|
|
1181
|
+
this.currentThinkingMessage.content += event.delta;
|
|
1182
|
+
}
|
|
1183
|
+
return [];
|
|
1184
|
+
}
|
|
1185
|
+
handleThinkingTextMessageEnd(event) {
|
|
1186
|
+
if (!this.currentThinkingMessage) {
|
|
1187
|
+
debug("Converter", "Thinking text end with no active thinking message");
|
|
1188
|
+
return [];
|
|
1189
|
+
}
|
|
1190
|
+
const content = this.currentThinkingMessage.content.trim();
|
|
1191
|
+
if (content.length === 0) {
|
|
1192
|
+
debug("Converter", "Skipping empty thinking message");
|
|
1193
|
+
this.currentThinkingMessage = null;
|
|
1194
|
+
return [];
|
|
1195
|
+
}
|
|
1196
|
+
const latencyMs = (event.timestamp || Date.now()) - this.currentThinkingMessage.startTime;
|
|
1197
|
+
const step = {
|
|
1198
|
+
id: uuidv4(),
|
|
1199
|
+
timestamp: this.currentThinkingMessage.startTime,
|
|
1200
|
+
type: "thinking",
|
|
1201
|
+
content,
|
|
1202
|
+
latencyMs
|
|
1203
|
+
};
|
|
1204
|
+
debug("Converter", `Thinking message completed: ${content.length} chars`);
|
|
1205
|
+
this.currentThinkingMessage = null;
|
|
1206
|
+
return [step];
|
|
1207
|
+
}
|
|
1208
|
+
extractToolName(title) {
|
|
1209
|
+
const runningMatch = title.match(/^Running\s+(.+)$/);
|
|
1210
|
+
if (runningMatch) return runningMatch[1];
|
|
1211
|
+
const completedMatch = title.match(/^(.+)\s+completed$/i);
|
|
1212
|
+
if (completedMatch) return completedMatch[1];
|
|
1213
|
+
return title;
|
|
1214
|
+
}
|
|
1215
|
+
parseToolArgs(description) {
|
|
1216
|
+
try {
|
|
1217
|
+
const args = {};
|
|
1218
|
+
const pairs = description.match(/(\w+):\s*("(?:[^"]|\\")*"|\w+)/g);
|
|
1219
|
+
if (pairs) {
|
|
1220
|
+
pairs.forEach((pair) => {
|
|
1221
|
+
const [key, rawValue] = pair.split(":").map((s) => s.trim());
|
|
1222
|
+
let value = rawValue;
|
|
1223
|
+
if (rawValue.startsWith('"') && rawValue.endsWith('"')) {
|
|
1224
|
+
value = rawValue.slice(1, -1);
|
|
1225
|
+
} else if (rawValue === "true") {
|
|
1226
|
+
value = true;
|
|
1227
|
+
} else if (rawValue === "false") {
|
|
1228
|
+
value = false;
|
|
1229
|
+
} else if (!isNaN(Number(rawValue))) {
|
|
1230
|
+
value = Number(rawValue);
|
|
1231
|
+
}
|
|
1232
|
+
args[key] = value;
|
|
1233
|
+
});
|
|
1234
|
+
return args;
|
|
1235
|
+
}
|
|
1236
|
+
return { description };
|
|
1237
|
+
} catch (e) {
|
|
1238
|
+
return { description };
|
|
1239
|
+
}
|
|
1240
|
+
}
|
|
1241
|
+
};
|
|
1242
|
+
function computeTrajectoryFromRawEvents(rawEvents) {
|
|
1243
|
+
const converter = new AGUIToTrajectoryConverter();
|
|
1244
|
+
const trajectory = [];
|
|
1245
|
+
for (const event of rawEvents) {
|
|
1246
|
+
const steps = converter.processEvent(event);
|
|
1247
|
+
trajectory.push(...steps);
|
|
1248
|
+
}
|
|
1249
|
+
trajectory.sort((a, b) => a.timestamp - b.timestamp);
|
|
1250
|
+
return trajectory;
|
|
1251
|
+
}
|
|
1252
|
+
|
|
1253
|
+
// services/connectors/agui/AGUIStreamingConnector.ts
|
|
1254
|
+
var AGUIStreamingConnector = class extends BaseConnector {
|
|
1255
|
+
constructor() {
|
|
1256
|
+
super(...arguments);
|
|
1257
|
+
this.type = "agui-streaming";
|
|
1258
|
+
this.name = "AG-UI Streaming";
|
|
1259
|
+
this.supportsStreaming = true;
|
|
1260
|
+
}
|
|
1261
|
+
/**
|
|
1262
|
+
* Build AG-UI payload from standard request
|
|
1263
|
+
*/
|
|
1264
|
+
buildPayload(request) {
|
|
1265
|
+
return buildAgentPayload(
|
|
1266
|
+
request.testCase,
|
|
1267
|
+
request.modelId,
|
|
1268
|
+
request.threadId,
|
|
1269
|
+
request.runId
|
|
1270
|
+
);
|
|
1271
|
+
}
|
|
1272
|
+
/**
|
|
1273
|
+
* Execute the request using SSE streaming
|
|
1274
|
+
*/
|
|
1275
|
+
async execute(endpoint, request, auth, onProgress, onRawEvent) {
|
|
1276
|
+
const hasPrebuiltPayload = !!request.payload;
|
|
1277
|
+
const payload = request.payload || this.buildPayload(request);
|
|
1278
|
+
const headers = this.buildAuthHeaders(auth);
|
|
1279
|
+
const trajectory = [];
|
|
1280
|
+
const rawEvents = [];
|
|
1281
|
+
const converter = new AGUIToTrajectoryConverter();
|
|
1282
|
+
this.debug("Executing AG-UI streaming request");
|
|
1283
|
+
await consumeSSEStream(
|
|
1284
|
+
endpoint,
|
|
1285
|
+
payload,
|
|
1286
|
+
(event) => {
|
|
1287
|
+
rawEvents.push(event);
|
|
1288
|
+
onRawEvent?.(event);
|
|
1289
|
+
const steps = converter.processEvent(event);
|
|
1290
|
+
steps.forEach((step) => {
|
|
1291
|
+
trajectory.push(step);
|
|
1292
|
+
onProgress?.(step);
|
|
1293
|
+
});
|
|
1294
|
+
},
|
|
1295
|
+
headers
|
|
1296
|
+
);
|
|
1297
|
+
const runId = converter.getRunId();
|
|
1298
|
+
this.debug("Stream completed. RunId:", runId, "Steps:", trajectory.length);
|
|
1299
|
+
return {
|
|
1300
|
+
trajectory,
|
|
1301
|
+
runId,
|
|
1302
|
+
rawEvents,
|
|
1303
|
+
metadata: {
|
|
1304
|
+
threadId: converter.getThreadId()
|
|
1305
|
+
}
|
|
1306
|
+
};
|
|
1307
|
+
}
|
|
1308
|
+
/**
|
|
1309
|
+
* Parse raw AG-UI events into trajectory steps
|
|
1310
|
+
* Used for re-processing stored raw events
|
|
1311
|
+
*/
|
|
1312
|
+
parseResponse(rawEvents) {
|
|
1313
|
+
return computeTrajectoryFromRawEvents(rawEvents);
|
|
1314
|
+
}
|
|
1315
|
+
/**
|
|
1316
|
+
* Health check for AG-UI endpoint
|
|
1317
|
+
* Tries to connect without sending a full request
|
|
1318
|
+
*/
|
|
1319
|
+
async healthCheck(endpoint, auth) {
|
|
1320
|
+
try {
|
|
1321
|
+
const headers = this.buildAuthHeaders(auth);
|
|
1322
|
+
const response = await fetch(endpoint, {
|
|
1323
|
+
method: "OPTIONS",
|
|
1324
|
+
headers
|
|
1325
|
+
});
|
|
1326
|
+
return true;
|
|
1327
|
+
} catch (error) {
|
|
1328
|
+
this.error("Health check failed:", error);
|
|
1329
|
+
return false;
|
|
1330
|
+
}
|
|
1331
|
+
}
|
|
1332
|
+
};
|
|
1333
|
+
var aguiStreamingConnector = new AGUIStreamingConnector();
|
|
1334
|
+
|
|
1335
|
+
// services/connectors/mock/MockConnector.ts
|
|
1336
|
+
var MockConnector = class extends BaseConnector {
|
|
1337
|
+
constructor() {
|
|
1338
|
+
super(...arguments);
|
|
1339
|
+
this.type = "mock";
|
|
1340
|
+
this.name = "Demo Agent (Mock)";
|
|
1341
|
+
this.supportsStreaming = true;
|
|
1342
|
+
}
|
|
1343
|
+
/**
|
|
1344
|
+
* Build payload - not used for mock but required by interface
|
|
1345
|
+
*/
|
|
1346
|
+
buildPayload(request) {
|
|
1347
|
+
return {
|
|
1348
|
+
question: request.testCase.initialPrompt,
|
|
1349
|
+
context: request.testCase.context
|
|
1350
|
+
};
|
|
1351
|
+
}
|
|
1352
|
+
/**
|
|
1353
|
+
* Execute mock request - generates realistic trajectory with delays
|
|
1354
|
+
*/
|
|
1355
|
+
async execute(endpoint, request, auth, onProgress, onRawEvent) {
|
|
1356
|
+
const trajectory = [];
|
|
1357
|
+
const rawEvents = [];
|
|
1358
|
+
const runId = `mock-run-${Date.now()}`;
|
|
1359
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
1360
|
+
this.debug("Generating mock trajectory for:", request.testCase.name);
|
|
1361
|
+
const emitStep = (step) => {
|
|
1362
|
+
trajectory.push(step);
|
|
1363
|
+
onProgress?.(step);
|
|
1364
|
+
rawEvents.push({ type: "MOCK_STEP", step });
|
|
1365
|
+
onRawEvent?.({ type: "MOCK_STEP", step });
|
|
1366
|
+
};
|
|
1367
|
+
await sleep(100);
|
|
1368
|
+
emitStep(this.createStep(
|
|
1369
|
+
"assistant",
|
|
1370
|
+
"I need to investigate this issue. Let me start by checking the cluster health and then drill down into specific metrics."
|
|
1371
|
+
));
|
|
1372
|
+
await sleep(300);
|
|
1373
|
+
emitStep(this.createStep(
|
|
1374
|
+
"action",
|
|
1375
|
+
"Calling opensearch_cluster_health...",
|
|
1376
|
+
{
|
|
1377
|
+
toolName: "opensearch_cluster_health",
|
|
1378
|
+
toolArgs: { local: true }
|
|
1379
|
+
}
|
|
1380
|
+
));
|
|
1381
|
+
await sleep(500);
|
|
1382
|
+
emitStep(this.createStep(
|
|
1383
|
+
"tool_result",
|
|
1384
|
+
JSON.stringify({ status: "yellow", number_of_nodes: 3, unassigned_shards: 0 }, null, 2),
|
|
1385
|
+
{
|
|
1386
|
+
status: "SUCCESS",
|
|
1387
|
+
latencyMs: 500
|
|
1388
|
+
}
|
|
1389
|
+
));
|
|
1390
|
+
await sleep(300);
|
|
1391
|
+
emitStep(this.createStep(
|
|
1392
|
+
"assistant",
|
|
1393
|
+
"The cluster is in yellow state. Let me check the node stats to identify which node might be causing issues."
|
|
1394
|
+
));
|
|
1395
|
+
await sleep(300);
|
|
1396
|
+
emitStep(this.createStep(
|
|
1397
|
+
"action",
|
|
1398
|
+
"Calling opensearch_nodes_stats...",
|
|
1399
|
+
{
|
|
1400
|
+
toolName: "opensearch_nodes_stats",
|
|
1401
|
+
toolArgs: { metric: "jvm,os" }
|
|
1402
|
+
}
|
|
1403
|
+
));
|
|
1404
|
+
await sleep(600);
|
|
1405
|
+
emitStep(this.createStep(
|
|
1406
|
+
"tool_result",
|
|
1407
|
+
"Node-1: CPU 12%, JVM Heap 45%\nNode-2: CPU 15%, JVM Heap 52%\nNode-3: CPU 98%, JVM Heap 89% (Data Node)",
|
|
1408
|
+
{
|
|
1409
|
+
status: "SUCCESS",
|
|
1410
|
+
latencyMs: 600
|
|
1411
|
+
}
|
|
1412
|
+
));
|
|
1413
|
+
await sleep(400);
|
|
1414
|
+
emitStep(this.createStep(
|
|
1415
|
+
"response",
|
|
1416
|
+
`## Root Cause Analysis Complete
|
|
1417
|
+
|
|
1418
|
+
**Finding:** High CPU utilization detected on Node-3 (98% CPU, 89% JVM Heap)
|
|
1419
|
+
|
|
1420
|
+
**Root Cause:** Node-3 is experiencing resource exhaustion, likely due to:
|
|
1421
|
+
1. Heavy indexing or search operations
|
|
1422
|
+
2. Garbage collection pressure from high heap usage
|
|
1423
|
+
3. Possible hot spot in shard distribution
|
|
1424
|
+
|
|
1425
|
+
**Recommendations:**
|
|
1426
|
+
1. Check hot threads on Node-3 using \`_nodes/Node-3/hot_threads\`
|
|
1427
|
+
2. Review shard distribution and consider rebalancing
|
|
1428
|
+
3. Monitor GC logs for long pauses
|
|
1429
|
+
4. Consider scaling horizontally if load persists`
|
|
1430
|
+
));
|
|
1431
|
+
this.debug("Mock trajectory completed. Steps:", trajectory.length);
|
|
1432
|
+
return {
|
|
1433
|
+
trajectory,
|
|
1434
|
+
runId,
|
|
1435
|
+
rawEvents,
|
|
1436
|
+
metadata: {
|
|
1437
|
+
mock: true,
|
|
1438
|
+
testCaseId: request.testCase.id,
|
|
1439
|
+
testCaseName: request.testCase.name
|
|
1440
|
+
}
|
|
1441
|
+
};
|
|
1442
|
+
}
|
|
1443
|
+
/**
|
|
1444
|
+
* Parse raw events - for mock, just extract steps from our custom format
|
|
1445
|
+
*/
|
|
1446
|
+
parseResponse(rawEvents) {
|
|
1447
|
+
return rawEvents.filter((e) => e.type === "MOCK_STEP" && e.step).map((e) => e.step);
|
|
1448
|
+
}
|
|
1449
|
+
/**
|
|
1450
|
+
* Health check - mock is always "healthy"
|
|
1451
|
+
*/
|
|
1452
|
+
async healthCheck(endpoint, auth) {
|
|
1453
|
+
return true;
|
|
1454
|
+
}
|
|
1455
|
+
};
|
|
1456
|
+
var mockConnector = new MockConnector();
|
|
1457
|
+
|
|
1458
|
+
// services/connectors/rest/RESTConnector.ts
|
|
1459
|
+
var RESTConnector = class extends BaseConnector {
|
|
1460
|
+
constructor() {
|
|
1461
|
+
super(...arguments);
|
|
1462
|
+
this.type = "rest";
|
|
1463
|
+
this.name = "REST API";
|
|
1464
|
+
this.supportsStreaming = false;
|
|
1465
|
+
}
|
|
1466
|
+
/**
|
|
1467
|
+
* Build generic REST payload
|
|
1468
|
+
* Can be customized via connectorConfig
|
|
1469
|
+
*/
|
|
1470
|
+
buildPayload(request) {
|
|
1471
|
+
return {
|
|
1472
|
+
prompt: request.testCase.initialPrompt,
|
|
1473
|
+
context: request.testCase.context,
|
|
1474
|
+
model: request.modelId,
|
|
1475
|
+
tools: request.testCase.tools
|
|
1476
|
+
};
|
|
1477
|
+
}
|
|
1478
|
+
/**
|
|
1479
|
+
* Execute REST request
|
|
1480
|
+
*/
|
|
1481
|
+
async execute(endpoint, request, auth, onProgress, onRawEvent) {
|
|
1482
|
+
const payload = request.payload || this.buildPayload(request);
|
|
1483
|
+
const headers = this.buildAuthHeaders(auth);
|
|
1484
|
+
this.debug("Executing REST request");
|
|
1485
|
+
this.debug("Endpoint:", endpoint);
|
|
1486
|
+
this.debug("Payload:", JSON.stringify(payload).substring(0, 500));
|
|
1487
|
+
const response = await fetch(endpoint, {
|
|
1488
|
+
method: "POST",
|
|
1489
|
+
headers: {
|
|
1490
|
+
"Content-Type": "application/json",
|
|
1491
|
+
...headers
|
|
1492
|
+
},
|
|
1493
|
+
body: JSON.stringify(payload)
|
|
1494
|
+
});
|
|
1495
|
+
if (!response.ok) {
|
|
1496
|
+
const errorText = await response.text();
|
|
1497
|
+
throw new Error(`REST request failed: ${response.status} - ${errorText}`);
|
|
1498
|
+
}
|
|
1499
|
+
const data = await response.json();
|
|
1500
|
+
onRawEvent?.(data);
|
|
1501
|
+
const trajectory = this.parseResponse(data);
|
|
1502
|
+
trajectory.forEach((step) => onProgress?.(step));
|
|
1503
|
+
return {
|
|
1504
|
+
trajectory,
|
|
1505
|
+
runId: data.runId || data.id || null,
|
|
1506
|
+
rawEvents: [data],
|
|
1507
|
+
metadata: {
|
|
1508
|
+
status: response.status,
|
|
1509
|
+
responseHeaders: Object.fromEntries(response.headers.entries())
|
|
1510
|
+
}
|
|
1511
|
+
};
|
|
1512
|
+
}
|
|
1513
|
+
/**
|
|
1514
|
+
* Parse REST response into trajectory steps
|
|
1515
|
+
* This is a generic implementation - subclass for specific APIs
|
|
1516
|
+
*/
|
|
1517
|
+
parseResponse(data) {
|
|
1518
|
+
const steps = [];
|
|
1519
|
+
if (data.thinking) {
|
|
1520
|
+
steps.push(this.createStep("thinking", data.thinking));
|
|
1521
|
+
}
|
|
1522
|
+
if (data.toolCalls && Array.isArray(data.toolCalls)) {
|
|
1523
|
+
for (const call of data.toolCalls) {
|
|
1524
|
+
steps.push(this.createStep("action", `Calling ${call.name}...`, {
|
|
1525
|
+
toolName: call.name,
|
|
1526
|
+
toolArgs: call.args || call.input
|
|
1527
|
+
}));
|
|
1528
|
+
if (call.result !== void 0) {
|
|
1529
|
+
steps.push(this.createStep(
|
|
1530
|
+
"tool_result",
|
|
1531
|
+
typeof call.result === "string" ? call.result : JSON.stringify(call.result),
|
|
1532
|
+
{ status: "SUCCESS" }
|
|
1533
|
+
));
|
|
1534
|
+
}
|
|
1535
|
+
}
|
|
1536
|
+
}
|
|
1537
|
+
const responseContent = data.response || data.content || data.answer || data.text || data.message;
|
|
1538
|
+
if (responseContent) {
|
|
1539
|
+
steps.push(this.createStep(
|
|
1540
|
+
"response",
|
|
1541
|
+
typeof responseContent === "string" ? responseContent : JSON.stringify(responseContent)
|
|
1542
|
+
));
|
|
1543
|
+
}
|
|
1544
|
+
if (data.inference_results) {
|
|
1545
|
+
const outputs = data.inference_results[0]?.output || [];
|
|
1546
|
+
for (const output of outputs) {
|
|
1547
|
+
if (output.name === "response") {
|
|
1548
|
+
const content = output.dataAsMap?.response || output.result;
|
|
1549
|
+
if (content) {
|
|
1550
|
+
steps.push(this.createStep("response", content));
|
|
1551
|
+
}
|
|
1552
|
+
}
|
|
1553
|
+
}
|
|
1554
|
+
}
|
|
1555
|
+
if (steps.length === 0 && data) {
|
|
1556
|
+
steps.push(this.createStep("response", JSON.stringify(data, null, 2)));
|
|
1557
|
+
}
|
|
1558
|
+
return steps;
|
|
1559
|
+
}
|
|
1560
|
+
};
|
|
1561
|
+
var restConnector = new RESTConnector();
|
|
1562
|
+
|
|
1563
|
+
// services/connectors/litellm/LiteLLMConnector.ts
|
|
1564
|
+
var LiteLLMConnector = class extends BaseConnector {
|
|
1565
|
+
constructor() {
|
|
1566
|
+
super(...arguments);
|
|
1567
|
+
this.type = "litellm";
|
|
1568
|
+
this.name = "LiteLLM / OpenAI-compatible";
|
|
1569
|
+
this.supportsStreaming = false;
|
|
1570
|
+
}
|
|
1571
|
+
/**
|
|
1572
|
+
* Build OpenAI Chat Completion payload from test case
|
|
1573
|
+
*/
|
|
1574
|
+
buildPayload(request) {
|
|
1575
|
+
const messages = [];
|
|
1576
|
+
if (request.testCase.context && request.testCase.context.length > 0) {
|
|
1577
|
+
const contextText = request.testCase.context.map((c) => typeof c === "string" ? c : JSON.stringify(c)).join("\n");
|
|
1578
|
+
messages.push({
|
|
1579
|
+
role: "system",
|
|
1580
|
+
content: contextText
|
|
1581
|
+
});
|
|
1582
|
+
}
|
|
1583
|
+
messages.push({
|
|
1584
|
+
role: "user",
|
|
1585
|
+
content: request.testCase.initialPrompt
|
|
1586
|
+
});
|
|
1587
|
+
const payload = {
|
|
1588
|
+
model: request.modelId,
|
|
1589
|
+
messages
|
|
1590
|
+
};
|
|
1591
|
+
if (request.testCase.tools && request.testCase.tools.length > 0) {
|
|
1592
|
+
payload.tools = request.testCase.tools.map((tool) => ({
|
|
1593
|
+
type: "function",
|
|
1594
|
+
function: {
|
|
1595
|
+
name: tool.name,
|
|
1596
|
+
description: tool.description || "",
|
|
1597
|
+
parameters: tool.parameters || {}
|
|
1598
|
+
}
|
|
1599
|
+
}));
|
|
1600
|
+
}
|
|
1601
|
+
return payload;
|
|
1602
|
+
}
|
|
1603
|
+
/**
|
|
1604
|
+
* Execute OpenAI-compatible Chat Completion request
|
|
1605
|
+
*/
|
|
1606
|
+
async execute(endpoint, request, auth, onProgress, onRawEvent) {
|
|
1607
|
+
const payload = request.payload || this.buildPayload(request);
|
|
1608
|
+
const headers = this.buildAuthHeaders(auth);
|
|
1609
|
+
this.debug("Executing LiteLLM request");
|
|
1610
|
+
this.debug("Endpoint:", endpoint);
|
|
1611
|
+
this.debug("Model:", payload.model);
|
|
1612
|
+
const response = await fetch(endpoint, {
|
|
1613
|
+
method: "POST",
|
|
1614
|
+
headers: {
|
|
1615
|
+
"Content-Type": "application/json",
|
|
1616
|
+
...headers
|
|
1617
|
+
},
|
|
1618
|
+
body: JSON.stringify(payload)
|
|
1619
|
+
});
|
|
1620
|
+
if (!response.ok) {
|
|
1621
|
+
const errorText = await response.text();
|
|
1622
|
+
throw new Error(`LiteLLM request failed: ${response.status} - ${errorText}`);
|
|
1623
|
+
}
|
|
1624
|
+
const data = await response.json();
|
|
1625
|
+
onRawEvent?.(data);
|
|
1626
|
+
const trajectory = this.parseResponse(data);
|
|
1627
|
+
trajectory.forEach((step) => onProgress?.(step));
|
|
1628
|
+
return {
|
|
1629
|
+
trajectory,
|
|
1630
|
+
runId: data.id || null,
|
|
1631
|
+
rawEvents: [data],
|
|
1632
|
+
metadata: {
|
|
1633
|
+
model: data.model,
|
|
1634
|
+
usage: data.usage,
|
|
1635
|
+
finishReason: data.choices?.[0]?.finish_reason
|
|
1636
|
+
}
|
|
1637
|
+
};
|
|
1638
|
+
}
|
|
1639
|
+
/**
|
|
1640
|
+
* Parse OpenAI Chat Completion response into trajectory steps
|
|
1641
|
+
*/
|
|
1642
|
+
parseResponse(data) {
|
|
1643
|
+
const steps = [];
|
|
1644
|
+
const choice = data.choices?.[0];
|
|
1645
|
+
if (!choice) {
|
|
1646
|
+
steps.push(this.createStep("response", JSON.stringify(data, null, 2)));
|
|
1647
|
+
return steps;
|
|
1648
|
+
}
|
|
1649
|
+
const message = choice.message;
|
|
1650
|
+
if (message.tool_calls && message.tool_calls.length > 0) {
|
|
1651
|
+
for (const toolCall of message.tool_calls) {
|
|
1652
|
+
let toolArgs;
|
|
1653
|
+
try {
|
|
1654
|
+
toolArgs = JSON.parse(toolCall.function.arguments);
|
|
1655
|
+
} catch {
|
|
1656
|
+
toolArgs = toolCall.function.arguments;
|
|
1657
|
+
}
|
|
1658
|
+
steps.push(this.createStep("action", `Calling ${toolCall.function.name}...`, {
|
|
1659
|
+
toolName: toolCall.function.name,
|
|
1660
|
+
toolArgs
|
|
1661
|
+
}));
|
|
1662
|
+
}
|
|
1663
|
+
}
|
|
1664
|
+
if (message.content) {
|
|
1665
|
+
steps.push(this.createStep("response", message.content));
|
|
1666
|
+
}
|
|
1667
|
+
if (steps.length === 0) {
|
|
1668
|
+
steps.push(this.createStep("response", "(empty response)"));
|
|
1669
|
+
}
|
|
1670
|
+
return steps;
|
|
1671
|
+
}
|
|
1672
|
+
};
|
|
1673
|
+
var litellmConnector = new LiteLLMConnector();
|
|
1674
|
+
|
|
1675
|
+
// services/connectors/index.ts
|
|
1676
|
+
connectorRegistry.register(aguiStreamingConnector);
|
|
1677
|
+
connectorRegistry.register(mockConnector);
|
|
1678
|
+
connectorRegistry.register(restConnector);
|
|
1679
|
+
connectorRegistry.register(litellmConnector);
|
|
1680
|
+
console.log("[Connectors] Browser-safe connectors registered:", connectorRegistry.getRegisteredTypes().join(", "));
|
|
1681
|
+
|
|
1682
|
+
// services/connectors/subprocess/SubprocessConnector.ts
|
|
1683
|
+
import { spawn } from "child_process";
|
|
1684
|
+
var DEFAULT_SUBPROCESS_CONFIG = {
|
|
1685
|
+
command: "",
|
|
1686
|
+
args: [],
|
|
1687
|
+
env: {},
|
|
1688
|
+
inputMode: "stdin",
|
|
1689
|
+
outputParser: "text",
|
|
1690
|
+
timeout: 3e5
|
|
1691
|
+
// 5 minutes
|
|
1692
|
+
};
|
|
1693
|
+
var SubprocessConnector = class extends BaseConnector {
|
|
1694
|
+
constructor(config) {
|
|
1695
|
+
super();
|
|
1696
|
+
this.type = "subprocess";
|
|
1697
|
+
this.name = "Subprocess (CLI)";
|
|
1698
|
+
this.supportsStreaming = true;
|
|
1699
|
+
this.config = { ...DEFAULT_SUBPROCESS_CONFIG, ...config };
|
|
1700
|
+
}
|
|
1701
|
+
/**
|
|
1702
|
+
* Build input for the subprocess
|
|
1703
|
+
*/
|
|
1704
|
+
buildPayload(request) {
|
|
1705
|
+
let prompt = request.testCase.initialPrompt;
|
|
1706
|
+
if (request.testCase.context && request.testCase.context.length > 0) {
|
|
1707
|
+
const contextStr = request.testCase.context.map((c) => `${c.description}: ${c.value}`).join("\n");
|
|
1708
|
+
prompt = `Context:
|
|
1709
|
+
${contextStr}
|
|
1710
|
+
|
|
1711
|
+
Question: ${prompt}`;
|
|
1712
|
+
}
|
|
1713
|
+
return prompt;
|
|
1714
|
+
}
|
|
1715
|
+
/**
|
|
1716
|
+
* Execute subprocess and capture output
|
|
1717
|
+
*/
|
|
1718
|
+
async execute(endpoint, request, auth, onProgress, onRawEvent) {
|
|
1719
|
+
this.debug("========== execute() STARTED ==========");
|
|
1720
|
+
const command = endpoint || this.config.command;
|
|
1721
|
+
const args = this.config.args || [];
|
|
1722
|
+
const input = request.payload || this.buildPayload(request);
|
|
1723
|
+
this.debug("Command:", command);
|
|
1724
|
+
this.debug("Args:", args);
|
|
1725
|
+
this.debug("Input mode:", this.config.inputMode);
|
|
1726
|
+
this.debug("Output parser:", this.config.outputParser);
|
|
1727
|
+
this.debug("Timeout:", this.config.timeout);
|
|
1728
|
+
this.debug("Input (first 500 chars):", input.substring(0, 500));
|
|
1729
|
+
this.debug("Working dir:", this.config.workingDir || process.cwd());
|
|
1730
|
+
const env = {
|
|
1731
|
+
...process.env,
|
|
1732
|
+
...this.buildAuthEnv(auth),
|
|
1733
|
+
...this.config.env
|
|
1734
|
+
};
|
|
1735
|
+
return new Promise((resolve5, reject) => {
|
|
1736
|
+
const trajectory = [];
|
|
1737
|
+
const rawOutput = [];
|
|
1738
|
+
let stdout = "";
|
|
1739
|
+
let stderr = "";
|
|
1740
|
+
let settled = false;
|
|
1741
|
+
const finalArgs = this.config.inputMode === "arg" ? [...args, input] : args;
|
|
1742
|
+
this.debug("Spawning process...");
|
|
1743
|
+
this.debug("Full command:", command, finalArgs.join(" "));
|
|
1744
|
+
const proc = spawn(command, finalArgs, {
|
|
1745
|
+
env,
|
|
1746
|
+
cwd: this.config.workingDir,
|
|
1747
|
+
shell: true
|
|
1748
|
+
});
|
|
1749
|
+
this.debug("Process spawned, PID:", proc.pid);
|
|
1750
|
+
const timeoutId = setTimeout(() => {
|
|
1751
|
+
if (settled) return;
|
|
1752
|
+
settled = true;
|
|
1753
|
+
this.debug("TIMEOUT reached, killing process");
|
|
1754
|
+
proc.kill("SIGTERM");
|
|
1755
|
+
reject(new Error(`Subprocess timed out after ${this.config.timeout}ms`));
|
|
1756
|
+
}, this.config.timeout);
|
|
1757
|
+
if (this.config.inputMode === "stdin") {
|
|
1758
|
+
this.debug("Writing input to stdin...");
|
|
1759
|
+
proc.stdin.write(input);
|
|
1760
|
+
proc.stdin.end();
|
|
1761
|
+
this.debug("stdin closed");
|
|
1762
|
+
}
|
|
1763
|
+
proc.stdout.on("data", (data) => {
|
|
1764
|
+
const chunk = data.toString();
|
|
1765
|
+
this.debug("stdout received:", chunk.length, "bytes");
|
|
1766
|
+
this.debug("stdout preview:", chunk.substring(0, 200));
|
|
1767
|
+
stdout += chunk;
|
|
1768
|
+
rawOutput.push({ type: "stdout", data: chunk, timestamp: Date.now() });
|
|
1769
|
+
onRawEvent?.({ type: "stdout", data: chunk });
|
|
1770
|
+
if (this.config.outputParser === "streaming") {
|
|
1771
|
+
this.parseStreamingOutput(chunk, trajectory, onProgress);
|
|
1772
|
+
}
|
|
1773
|
+
});
|
|
1774
|
+
proc.stderr.on("data", (data) => {
|
|
1775
|
+
const chunk = data.toString();
|
|
1776
|
+
this.debug("stderr received:", chunk.length, "bytes");
|
|
1777
|
+
this.debug("stderr:", chunk);
|
|
1778
|
+
stderr += chunk;
|
|
1779
|
+
onRawEvent?.({ type: "stderr", data: chunk });
|
|
1780
|
+
this.debug("stderr:", chunk);
|
|
1781
|
+
});
|
|
1782
|
+
proc.on("close", (code, signal) => {
|
|
1783
|
+
this.debug("Process closed with code:", code, "signal:", signal);
|
|
1784
|
+
clearTimeout(timeoutId);
|
|
1785
|
+
if (settled) return;
|
|
1786
|
+
settled = true;
|
|
1787
|
+
if (code !== 0) {
|
|
1788
|
+
this.debug("Non-zero exit code:", code);
|
|
1789
|
+
this.error(`Process exited with code ${code}`);
|
|
1790
|
+
this.error("stderr:", stderr);
|
|
1791
|
+
}
|
|
1792
|
+
const finalTrajectory = this.config.outputParser === "streaming" ? trajectory : this.parseResponse({ stdout, stderr, exitCode: code });
|
|
1793
|
+
if (this.config.outputParser !== "streaming") {
|
|
1794
|
+
finalTrajectory.forEach((step) => onProgress?.(step));
|
|
1795
|
+
}
|
|
1796
|
+
this.debug("Resolving with trajectory of", finalTrajectory.length, "steps");
|
|
1797
|
+
resolve5({
|
|
1798
|
+
trajectory: finalTrajectory,
|
|
1799
|
+
runId: `subprocess-${Date.now()}`,
|
|
1800
|
+
rawEvents: rawOutput,
|
|
1801
|
+
metadata: {
|
|
1802
|
+
command,
|
|
1803
|
+
args: finalArgs,
|
|
1804
|
+
exitCode: code,
|
|
1805
|
+
stderr: stderr || void 0
|
|
1806
|
+
}
|
|
1807
|
+
});
|
|
1808
|
+
});
|
|
1809
|
+
proc.on("error", (error) => {
|
|
1810
|
+
this.debug("ERROR event:", error.message);
|
|
1811
|
+
clearTimeout(timeoutId);
|
|
1812
|
+
if (settled) return;
|
|
1813
|
+
settled = true;
|
|
1814
|
+
let errorMsg = `Failed to spawn subprocess: ${error.message}`;
|
|
1815
|
+
if (error.message.includes("ENOENT")) {
|
|
1816
|
+
errorMsg = `Command '${command}' not found. Is it installed and in PATH?`;
|
|
1817
|
+
console.error(`[Subprocess] ENOENT error - command '${command}' not found in PATH`);
|
|
1818
|
+
} else if (error.message.includes("EACCES")) {
|
|
1819
|
+
errorMsg = `Permission denied executing '${command}'. Check file permissions.`;
|
|
1820
|
+
console.error(`[Subprocess] EACCES error - permission denied for '${command}'`);
|
|
1821
|
+
} else if (error.message.includes("EPERM")) {
|
|
1822
|
+
errorMsg = `Operation not permitted for '${command}'. May require elevated privileges.`;
|
|
1823
|
+
console.error(`[Subprocess] EPERM error - operation not permitted`);
|
|
1824
|
+
}
|
|
1825
|
+
reject(new Error(errorMsg));
|
|
1826
|
+
});
|
|
1827
|
+
});
|
|
1828
|
+
this.debug("========== execute() COMPLETED ==========");
|
|
1829
|
+
}
|
|
1830
|
+
/**
|
|
1831
|
+
* Parse streaming output and emit steps in real-time
|
|
1832
|
+
*/
|
|
1833
|
+
parseStreamingOutput(chunk, trajectory, onProgress) {
|
|
1834
|
+
const lines = chunk.split("\n").filter((line) => line.trim());
|
|
1835
|
+
for (const line of lines) {
|
|
1836
|
+
const step = this.createStep("assistant", line);
|
|
1837
|
+
trajectory.push(step);
|
|
1838
|
+
onProgress?.(step);
|
|
1839
|
+
}
|
|
1840
|
+
}
|
|
1841
|
+
/**
|
|
1842
|
+
* Parse final subprocess output
|
|
1843
|
+
*/
|
|
1844
|
+
parseResponse(data) {
|
|
1845
|
+
const steps = [];
|
|
1846
|
+
if (this.config.outputParser === "json") {
|
|
1847
|
+
try {
|
|
1848
|
+
const parsed = JSON.parse(data.stdout);
|
|
1849
|
+
return this.parseJsonOutput(parsed);
|
|
1850
|
+
} catch {
|
|
1851
|
+
this.debug("Failed to parse JSON output, falling back to text");
|
|
1852
|
+
}
|
|
1853
|
+
}
|
|
1854
|
+
if (data.stdout.trim()) {
|
|
1855
|
+
steps.push(this.createStep("response", data.stdout.trim()));
|
|
1856
|
+
}
|
|
1857
|
+
if (data.exitCode !== 0 && data.stderr.trim()) {
|
|
1858
|
+
steps.push(this.createStep("tool_result", `Error: ${data.stderr.trim()}`, {
|
|
1859
|
+
status: "FAILURE"
|
|
1860
|
+
}));
|
|
1861
|
+
}
|
|
1862
|
+
return steps;
|
|
1863
|
+
}
|
|
1864
|
+
/**
|
|
1865
|
+
* Parse JSON output into trajectory steps
|
|
1866
|
+
*/
|
|
1867
|
+
parseJsonOutput(data) {
|
|
1868
|
+
const steps = [];
|
|
1869
|
+
if (data.thinking) {
|
|
1870
|
+
steps.push(this.createStep("thinking", data.thinking));
|
|
1871
|
+
}
|
|
1872
|
+
if (data.steps && Array.isArray(data.steps)) {
|
|
1873
|
+
for (const step of data.steps) {
|
|
1874
|
+
steps.push(this.createStep(step.type || "assistant", step.content, {
|
|
1875
|
+
toolName: step.toolName,
|
|
1876
|
+
toolArgs: step.toolArgs
|
|
1877
|
+
}));
|
|
1878
|
+
}
|
|
1879
|
+
}
|
|
1880
|
+
if (data.response || data.answer || data.content) {
|
|
1881
|
+
steps.push(this.createStep("response", data.response || data.answer || data.content));
|
|
1882
|
+
}
|
|
1883
|
+
return steps;
|
|
1884
|
+
}
|
|
1885
|
+
/**
|
|
1886
|
+
* Health check - verify command exists
|
|
1887
|
+
*/
|
|
1888
|
+
async healthCheck(endpoint, auth) {
|
|
1889
|
+
const command = endpoint || this.config.command;
|
|
1890
|
+
if (!command) return false;
|
|
1891
|
+
return new Promise((resolve5) => {
|
|
1892
|
+
const proc = spawn("which", [command], { shell: true });
|
|
1893
|
+
proc.on("close", (code) => resolve5(code === 0));
|
|
1894
|
+
proc.on("error", () => resolve5(false));
|
|
1895
|
+
});
|
|
1896
|
+
}
|
|
1897
|
+
};
|
|
1898
|
+
var subprocessConnector = new SubprocessConnector();
|
|
1899
|
+
|
|
1900
|
+
// services/connectors/claude-code/ClaudeCodeConnector.ts
|
|
1901
|
+
var CLAUDE_CODE_DEFAULT_CONFIG = {
|
|
1902
|
+
command: "claude",
|
|
1903
|
+
args: ["--print", "--verbose", "--output-format", "stream-json"],
|
|
1904
|
+
// Structured JSON output (--verbose required with stream-json)
|
|
1905
|
+
env: {
|
|
1906
|
+
// These can be overridden by agent config or environment
|
|
1907
|
+
DISABLE_PROMPT_CACHING: "1",
|
|
1908
|
+
DISABLE_ERROR_REPORTING: "1"
|
|
1909
|
+
// Note: DISABLE_TELEMETRY removed - telemetry enabled by default
|
|
1910
|
+
// Configure OTEL_EXPORTER_OTLP_ENDPOINT in .env to send traces
|
|
1911
|
+
},
|
|
1912
|
+
inputMode: "stdin",
|
|
1913
|
+
outputParser: "streaming",
|
|
1914
|
+
timeout: 6e5
|
|
1915
|
+
// 10 minutes for Claude Code
|
|
1916
|
+
};
|
|
1917
|
+
var ClaudeCodeConnector = class extends SubprocessConnector {
|
|
1918
|
+
constructor(config) {
|
|
1919
|
+
super({ ...CLAUDE_CODE_DEFAULT_CONFIG, ...config });
|
|
1920
|
+
this.type = "claude-code";
|
|
1921
|
+
this.name = "Claude Code CLI";
|
|
1922
|
+
this.outputBuffer = "";
|
|
1923
|
+
this.thinkingBuffer = "";
|
|
1924
|
+
this.isInThinking = false;
|
|
1925
|
+
}
|
|
1926
|
+
/**
|
|
1927
|
+
* Build prompt for Claude Code
|
|
1928
|
+
* Structures the input to get the best RCA results
|
|
1929
|
+
*/
|
|
1930
|
+
buildPayload(request) {
|
|
1931
|
+
const parts = [];
|
|
1932
|
+
if (request.testCase.context && request.testCase.context.length > 0) {
|
|
1933
|
+
parts.push("## Context");
|
|
1934
|
+
for (const ctx of request.testCase.context) {
|
|
1935
|
+
parts.push(`**${ctx.description}:**`);
|
|
1936
|
+
parts.push(ctx.value);
|
|
1937
|
+
parts.push("");
|
|
1938
|
+
}
|
|
1939
|
+
}
|
|
1940
|
+
parts.push("## Task");
|
|
1941
|
+
parts.push(request.testCase.initialPrompt);
|
|
1942
|
+
return parts.join("\n");
|
|
1943
|
+
}
|
|
1944
|
+
/**
|
|
1945
|
+
* Parse Claude Code streaming output (stream-json format)
|
|
1946
|
+
* Each line is a JSON object with type and content
|
|
1947
|
+
*/
|
|
1948
|
+
parseStreamingOutput(chunk, trajectory, onProgress) {
|
|
1949
|
+
this.outputBuffer += chunk;
|
|
1950
|
+
const lines = this.outputBuffer.split("\n");
|
|
1951
|
+
this.outputBuffer = lines.pop() || "";
|
|
1952
|
+
for (const line of lines) {
|
|
1953
|
+
const trimmed = line.trim();
|
|
1954
|
+
if (!trimmed) continue;
|
|
1955
|
+
try {
|
|
1956
|
+
const event = JSON.parse(trimmed);
|
|
1957
|
+
const steps = this.parseJsonEvent(event);
|
|
1958
|
+
for (const step of steps) {
|
|
1959
|
+
trajectory.push(step);
|
|
1960
|
+
onProgress?.(step);
|
|
1961
|
+
}
|
|
1962
|
+
} catch {
|
|
1963
|
+
if (trimmed) {
|
|
1964
|
+
const step = this.createStep("assistant", trimmed);
|
|
1965
|
+
trajectory.push(step);
|
|
1966
|
+
onProgress?.(step);
|
|
1967
|
+
}
|
|
1968
|
+
}
|
|
1969
|
+
}
|
|
1970
|
+
}
|
|
1971
|
+
/**
|
|
1972
|
+
* Parse a single JSON event from stream-json output
|
|
1973
|
+
*/
|
|
1974
|
+
parseJsonEvent(event) {
|
|
1975
|
+
const steps = [];
|
|
1976
|
+
if (event.type === "assistant" && event.message?.content) {
|
|
1977
|
+
for (const block of event.message.content) {
|
|
1978
|
+
if (block.type === "thinking" && block.thinking) {
|
|
1979
|
+
steps.push(this.createStep("thinking", block.thinking));
|
|
1980
|
+
} else if (block.type === "text" && block.text) {
|
|
1981
|
+
steps.push(this.createStep("assistant", block.text));
|
|
1982
|
+
} else if (block.type === "tool_use") {
|
|
1983
|
+
steps.push(this.createStep("action", JSON.stringify(block.input || {}), {
|
|
1984
|
+
toolName: block.name,
|
|
1985
|
+
toolArgs: block.input
|
|
1986
|
+
}));
|
|
1987
|
+
}
|
|
1988
|
+
}
|
|
1989
|
+
} else if (event.type === "content_block_delta") {
|
|
1990
|
+
if (event.delta?.type === "thinking_delta" && event.delta.thinking) {
|
|
1991
|
+
this.thinkingBuffer += event.delta.thinking;
|
|
1992
|
+
} else if (event.delta?.type === "text_delta" && event.delta.text) {
|
|
1993
|
+
steps.push(this.createStep("assistant", event.delta.text));
|
|
1994
|
+
}
|
|
1995
|
+
} else if (event.type === "content_block_stop" && this.thinkingBuffer) {
|
|
1996
|
+
steps.push(this.createStep("thinking", this.thinkingBuffer));
|
|
1997
|
+
this.thinkingBuffer = "";
|
|
1998
|
+
} else if (event.type === "result" && event.result) {
|
|
1999
|
+
steps.push(this.createStep(
|
|
2000
|
+
"response",
|
|
2001
|
+
typeof event.result === "string" ? event.result : JSON.stringify(event.result)
|
|
2002
|
+
));
|
|
2003
|
+
} else if (event.type === "tool_result") {
|
|
2004
|
+
steps.push(this.createStep(
|
|
2005
|
+
"tool_result",
|
|
2006
|
+
typeof event.content === "string" ? event.content : JSON.stringify(event.content),
|
|
2007
|
+
{ status: event.is_error ? "FAILURE" /* FAILURE */ : "SUCCESS" /* SUCCESS */ }
|
|
2008
|
+
));
|
|
2009
|
+
}
|
|
2010
|
+
return steps;
|
|
2011
|
+
}
|
|
2012
|
+
/**
|
|
2013
|
+
* Parse final output for Claude Code
|
|
2014
|
+
*/
|
|
2015
|
+
parseResponse(data) {
|
|
2016
|
+
const steps = [];
|
|
2017
|
+
let content = data.stdout;
|
|
2018
|
+
const thinkingMatches = content.matchAll(/<thinking>([\s\S]*?)<\/thinking>/g);
|
|
2019
|
+
for (const match of thinkingMatches) {
|
|
2020
|
+
const thinking = match[1].trim();
|
|
2021
|
+
if (thinking) {
|
|
2022
|
+
steps.push(this.createStep("thinking", thinking));
|
|
2023
|
+
}
|
|
2024
|
+
content = content.replace(match[0], "");
|
|
2025
|
+
}
|
|
2026
|
+
const response = content.trim();
|
|
2027
|
+
if (response) {
|
|
2028
|
+
steps.push(this.createStep("response", response));
|
|
2029
|
+
}
|
|
2030
|
+
if (data.exitCode !== 0 && data.stderr.trim()) {
|
|
2031
|
+
steps.push(this.createStep("tool_result", `Error: ${data.stderr.trim()}`, {
|
|
2032
|
+
status: "FAILURE" /* FAILURE */
|
|
2033
|
+
}));
|
|
2034
|
+
}
|
|
2035
|
+
return steps;
|
|
2036
|
+
}
|
|
2037
|
+
/**
|
|
2038
|
+
* Reset state for new execution
|
|
2039
|
+
*/
|
|
2040
|
+
resetState() {
|
|
2041
|
+
this.outputBuffer = "";
|
|
2042
|
+
this.thinkingBuffer = "";
|
|
2043
|
+
this.isInThinking = false;
|
|
2044
|
+
}
|
|
2045
|
+
/**
|
|
2046
|
+
* Override execute to reset state
|
|
2047
|
+
*/
|
|
2048
|
+
async execute(endpoint, request, auth, onProgress, onRawEvent) {
|
|
2049
|
+
this.debug("========== execute() STARTED ==========");
|
|
2050
|
+
this.debug("Endpoint:", endpoint);
|
|
2051
|
+
this.debug("Test case:", request.testCase.name);
|
|
2052
|
+
this.debug("Config:", this["config"]);
|
|
2053
|
+
this.resetState();
|
|
2054
|
+
this.debug("State reset, calling super.execute()...");
|
|
2055
|
+
const result = await super.execute(endpoint, request, auth, onProgress, onRawEvent);
|
|
2056
|
+
this.debug("super.execute() returned with", result.trajectory.length, "steps");
|
|
2057
|
+
this.debug("========== execute() COMPLETED ==========");
|
|
2058
|
+
return result;
|
|
2059
|
+
}
|
|
2060
|
+
/**
|
|
2061
|
+
* Health check - verify claude command exists
|
|
2062
|
+
*/
|
|
2063
|
+
async healthCheck(endpoint, auth) {
|
|
2064
|
+
return super.healthCheck(endpoint || "claude", auth);
|
|
2065
|
+
}
|
|
2066
|
+
};
|
|
2067
|
+
var claudeCodeConnector = new ClaudeCodeConnector();
|
|
2068
|
+
|
|
2069
|
+
// services/connectors/server.ts
|
|
2070
|
+
connectorRegistry.register(subprocessConnector);
|
|
2071
|
+
connectorRegistry.register(claudeCodeConnector);
|
|
2072
|
+
console.log("[Connectors] Server connectors registered:", connectorRegistry.getRegisteredTypes().join(", "));
|
|
2073
|
+
|
|
2074
|
+
// cli/utils/serverLifecycle.ts
|
|
2075
|
+
import { spawn as spawn2, execSync } from "child_process";
|
|
2076
|
+
import net from "net";
|
|
2077
|
+
import { readFileSync } from "fs";
|
|
2078
|
+
import { fileURLToPath as fileURLToPath2 } from "url";
|
|
2079
|
+
import { dirname as dirname2, join as join3 } from "path";
|
|
2080
|
+
var __filename2 = fileURLToPath2(import.meta.url);
|
|
2081
|
+
var __dirname2 = dirname2(__filename2);
|
|
2082
|
+
var packageJsonPath = join3(__dirname2, "..", "..", "package.json");
|
|
2083
|
+
var cachedVersion = null;
|
|
2084
|
+
function getCliVersion() {
|
|
2085
|
+
if (cachedVersion !== null) {
|
|
2086
|
+
return cachedVersion;
|
|
2087
|
+
}
|
|
2088
|
+
try {
|
|
2089
|
+
const packageJson = JSON.parse(readFileSync(packageJsonPath, "utf-8"));
|
|
2090
|
+
cachedVersion = packageJson.version || "unknown";
|
|
2091
|
+
} catch {
|
|
2092
|
+
try {
|
|
2093
|
+
const altPath = join3(__dirname2, "..", "..", "..", "package.json");
|
|
2094
|
+
const packageJson = JSON.parse(readFileSync(altPath, "utf-8"));
|
|
2095
|
+
cachedVersion = packageJson.version || "unknown";
|
|
2096
|
+
} catch {
|
|
2097
|
+
cachedVersion = "unknown";
|
|
2098
|
+
}
|
|
2099
|
+
}
|
|
2100
|
+
return cachedVersion;
|
|
2101
|
+
}
|
|
2102
|
+
async function isServerRunning(port) {
|
|
2103
|
+
const controller = new AbortController();
|
|
2104
|
+
const timeout = setTimeout(() => controller.abort(), 2e3);
|
|
2105
|
+
try {
|
|
2106
|
+
const response = await fetch(`http://localhost:${port}/health`, {
|
|
2107
|
+
signal: controller.signal
|
|
2108
|
+
});
|
|
2109
|
+
if (response.ok) {
|
|
2110
|
+
return true;
|
|
2111
|
+
}
|
|
2112
|
+
} catch {
|
|
2113
|
+
} finally {
|
|
2114
|
+
clearTimeout(timeout);
|
|
2115
|
+
}
|
|
2116
|
+
return new Promise((resolve5) => {
|
|
2117
|
+
const socket = new net.Socket();
|
|
2118
|
+
socket.setTimeout(1e3);
|
|
2119
|
+
socket.on("connect", () => {
|
|
2120
|
+
socket.destroy();
|
|
2121
|
+
resolve5(true);
|
|
2122
|
+
});
|
|
2123
|
+
socket.on("timeout", () => {
|
|
2124
|
+
socket.destroy();
|
|
2125
|
+
resolve5(false);
|
|
2126
|
+
});
|
|
2127
|
+
socket.on("error", () => {
|
|
2128
|
+
resolve5(false);
|
|
2129
|
+
});
|
|
2130
|
+
socket.connect(port, "localhost");
|
|
2131
|
+
});
|
|
2132
|
+
}
|
|
2133
|
+
async function checkServerStatus(port) {
|
|
2134
|
+
const controller = new AbortController();
|
|
2135
|
+
const timeout = setTimeout(() => controller.abort(), 2e3);
|
|
2136
|
+
try {
|
|
2137
|
+
const response = await fetch(`http://localhost:${port}/health`, {
|
|
2138
|
+
signal: controller.signal
|
|
2139
|
+
});
|
|
2140
|
+
if (response.ok) {
|
|
2141
|
+
const data = await response.json();
|
|
2142
|
+
return {
|
|
2143
|
+
running: true,
|
|
2144
|
+
version: data.version
|
|
2145
|
+
};
|
|
2146
|
+
}
|
|
2147
|
+
} catch {
|
|
2148
|
+
} finally {
|
|
2149
|
+
clearTimeout(timeout);
|
|
2150
|
+
}
|
|
2151
|
+
return { running: false };
|
|
2152
|
+
}
|
|
2153
|
+
async function killServerOnPort(port) {
|
|
2154
|
+
try {
|
|
2155
|
+
if (process.platform !== "win32") {
|
|
2156
|
+
try {
|
|
2157
|
+
execSync(`lsof -t -i:${port} -sTCP:LISTEN | xargs kill -9 2>/dev/null || true`, { stdio: "ignore" });
|
|
2158
|
+
} catch {
|
|
2159
|
+
}
|
|
2160
|
+
} else {
|
|
2161
|
+
try {
|
|
2162
|
+
const result = execSync(`netstat -ano | findstr :${port}`, { encoding: "utf-8" });
|
|
2163
|
+
const lines = result.trim().split("\n");
|
|
2164
|
+
for (const line of lines) {
|
|
2165
|
+
const parts = line.trim().split(/\s+/);
|
|
2166
|
+
const pid = parts[parts.length - 1];
|
|
2167
|
+
if (pid && !isNaN(parseInt(pid))) {
|
|
2168
|
+
try {
|
|
2169
|
+
execSync(`taskkill /PID ${pid} /F`, { stdio: "ignore" });
|
|
2170
|
+
} catch {
|
|
2171
|
+
}
|
|
2172
|
+
}
|
|
2173
|
+
}
|
|
2174
|
+
} catch {
|
|
2175
|
+
}
|
|
2176
|
+
}
|
|
2177
|
+
const maxRetries = 10;
|
|
2178
|
+
const retryDelay = 500;
|
|
2179
|
+
for (let i = 0; i < maxRetries; i++) {
|
|
2180
|
+
await new Promise((r) => setTimeout(r, retryDelay));
|
|
2181
|
+
const stillRunning = await isServerRunning(port);
|
|
2182
|
+
if (!stillRunning) {
|
|
2183
|
+
return;
|
|
2184
|
+
}
|
|
2185
|
+
}
|
|
2186
|
+
console.warn(`[ServerLifecycle] Port ${port} may still be in use after ${maxRetries} retries`);
|
|
2187
|
+
} catch {
|
|
2188
|
+
}
|
|
2189
|
+
}
|
|
2190
|
+
async function waitForServer(port, timeout) {
|
|
2191
|
+
const startTime = Date.now();
|
|
2192
|
+
const pollInterval = 500;
|
|
2193
|
+
while (Date.now() - startTime < timeout) {
|
|
2194
|
+
if (await isServerRunning(port)) {
|
|
2195
|
+
return true;
|
|
2196
|
+
}
|
|
2197
|
+
await new Promise((r) => setTimeout(r, pollInterval));
|
|
2198
|
+
}
|
|
2199
|
+
return false;
|
|
2200
|
+
}
|
|
2201
|
+
async function startServer2(port, timeout) {
|
|
2202
|
+
const packageRoot = join3(__dirname2, "..", "..");
|
|
2203
|
+
const cliPath = join3(packageRoot, "bin", "cli.js");
|
|
2204
|
+
const child = spawn2("node", [cliPath, "serve", "-p", String(port), "--no-browser"], {
|
|
2205
|
+
detached: true,
|
|
2206
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
2207
|
+
env: {
|
|
2208
|
+
...process.env
|
|
2209
|
+
}
|
|
2210
|
+
});
|
|
2211
|
+
let stderrOutput = "";
|
|
2212
|
+
let stdoutOutput = "";
|
|
2213
|
+
child.stderr?.on("data", (data) => {
|
|
2214
|
+
stderrOutput += data.toString();
|
|
2215
|
+
});
|
|
2216
|
+
child.stdout?.on("data", (data) => {
|
|
2217
|
+
stdoutOutput += data.toString();
|
|
2218
|
+
});
|
|
2219
|
+
let earlyExit = false;
|
|
2220
|
+
let exitCode = null;
|
|
2221
|
+
child.on("exit", (code) => {
|
|
2222
|
+
earlyExit = true;
|
|
2223
|
+
exitCode = code;
|
|
2224
|
+
});
|
|
2225
|
+
child.unref();
|
|
2226
|
+
const ready = await waitForServer(port, timeout);
|
|
2227
|
+
if (!ready) {
|
|
2228
|
+
try {
|
|
2229
|
+
child.kill();
|
|
2230
|
+
} catch {
|
|
2231
|
+
}
|
|
2232
|
+
if (earlyExit) {
|
|
2233
|
+
console.error(`[ServerLifecycle] Server process exited with code ${exitCode} before becoming ready`);
|
|
2234
|
+
} else {
|
|
2235
|
+
console.error(`[ServerLifecycle] Server process did not respond to health checks within ${timeout}ms`);
|
|
2236
|
+
}
|
|
2237
|
+
if (stderrOutput) {
|
|
2238
|
+
console.error(`[ServerLifecycle] Server stderr:
|
|
2239
|
+
${stderrOutput}`);
|
|
2240
|
+
}
|
|
2241
|
+
if (stdoutOutput) {
|
|
2242
|
+
console.error(`[ServerLifecycle] Server stdout:
|
|
2243
|
+
${stdoutOutput}`);
|
|
2244
|
+
}
|
|
2245
|
+
if (!stderrOutput && !stdoutOutput) {
|
|
2246
|
+
console.error(`[ServerLifecycle] No output captured from server process`);
|
|
2247
|
+
console.error(`[ServerLifecycle] CLI path: ${cliPath}`);
|
|
2248
|
+
console.error(`[ServerLifecycle] Package root: ${packageRoot}`);
|
|
2249
|
+
}
|
|
2250
|
+
throw new Error(`Server failed to start within ${timeout}ms on port ${port}`);
|
|
2251
|
+
}
|
|
2252
|
+
return child;
|
|
2253
|
+
}
|
|
2254
|
+
function stopServer(process2) {
|
|
2255
|
+
try {
|
|
2256
|
+
if (process2.pid) {
|
|
2257
|
+
try {
|
|
2258
|
+
process2.kill("SIGTERM");
|
|
2259
|
+
} catch {
|
|
2260
|
+
}
|
|
2261
|
+
}
|
|
2262
|
+
} catch {
|
|
2263
|
+
}
|
|
2264
|
+
}
|
|
2265
|
+
async function ensureServer(config) {
|
|
2266
|
+
const { port, reuseExistingServer, startTimeout } = config;
|
|
2267
|
+
const baseUrl = `http://localhost:${port}`;
|
|
2268
|
+
const serverStatus = await checkServerStatus(port);
|
|
2269
|
+
const cliVersion = getCliVersion();
|
|
2270
|
+
if (serverStatus.running) {
|
|
2271
|
+
const versionMatches = serverStatus.version === cliVersion || serverStatus.version === "unknown" || cliVersion === "unknown";
|
|
2272
|
+
if (!versionMatches) {
|
|
2273
|
+
console.log(`[ServerLifecycle] Version mismatch detected!`);
|
|
2274
|
+
console.log(`[ServerLifecycle] Server version: ${serverStatus.version}`);
|
|
2275
|
+
console.log(`[ServerLifecycle] CLI version: ${cliVersion}`);
|
|
2276
|
+
if (reuseExistingServer) {
|
|
2277
|
+
console.log(`[ServerLifecycle] Stopping old server and starting v${cliVersion}...`);
|
|
2278
|
+
await killServerOnPort(port);
|
|
2279
|
+
} else {
|
|
2280
|
+
throw new Error(
|
|
2281
|
+
`Server version mismatch: server=${serverStatus.version}, CLI=${cliVersion}. Stop the existing server or upgrade to matching version.`
|
|
2282
|
+
);
|
|
2283
|
+
}
|
|
2284
|
+
} else if (reuseExistingServer) {
|
|
2285
|
+
console.log(`[ServerLifecycle] Reusing existing server (version ${serverStatus.version})`);
|
|
2286
|
+
return {
|
|
2287
|
+
wasStarted: false,
|
|
2288
|
+
baseUrl
|
|
2289
|
+
};
|
|
2290
|
+
} else {
|
|
2291
|
+
throw new Error(
|
|
2292
|
+
`Server already running on port ${port}. In CI mode (reuseExistingServer=false), this is an error. Stop the existing server or set reuseExistingServer: true.`
|
|
2293
|
+
);
|
|
2294
|
+
}
|
|
2295
|
+
}
|
|
2296
|
+
const serverProcess = await startServer2(port, startTimeout);
|
|
2297
|
+
return {
|
|
2298
|
+
wasStarted: true,
|
|
2299
|
+
baseUrl,
|
|
2300
|
+
process: serverProcess
|
|
2301
|
+
};
|
|
2302
|
+
}
|
|
2303
|
+
function createServerCleanup(result, isCI) {
|
|
2304
|
+
return () => {
|
|
2305
|
+
if (result.wasStarted && isCI && result.process) {
|
|
2306
|
+
stopServer(result.process);
|
|
2307
|
+
}
|
|
2308
|
+
};
|
|
2309
|
+
}
|
|
2310
|
+
|
|
2311
|
+
// cli/utils/apiClient.ts
|
|
2312
|
+
var ApiClient = class {
|
|
2313
|
+
constructor(baseUrl) {
|
|
2314
|
+
this.baseUrl = baseUrl;
|
|
2315
|
+
}
|
|
2316
|
+
/**
|
|
2317
|
+
* Check if server is healthy, with optional retries and exponential backoff.
|
|
2318
|
+
*/
|
|
2319
|
+
async checkHealth(retries = 2, delayMs = 500) {
|
|
2320
|
+
let lastError;
|
|
2321
|
+
for (let attempt = 0; attempt <= retries; attempt++) {
|
|
2322
|
+
try {
|
|
2323
|
+
const res = await fetch(`${this.baseUrl}/health`);
|
|
2324
|
+
if (!res.ok) {
|
|
2325
|
+
throw new Error(`Server health check failed: ${res.status} ${res.statusText}`);
|
|
2326
|
+
}
|
|
2327
|
+
return res.json();
|
|
2328
|
+
} catch (err) {
|
|
2329
|
+
lastError = err instanceof Error ? err : new Error(String(err));
|
|
2330
|
+
if (attempt < retries) {
|
|
2331
|
+
await new Promise((resolve5) => setTimeout(resolve5, delayMs * Math.pow(2, attempt)));
|
|
2332
|
+
}
|
|
2333
|
+
}
|
|
2334
|
+
}
|
|
2335
|
+
throw lastError;
|
|
2336
|
+
}
|
|
2337
|
+
/**
|
|
2338
|
+
* List all benchmarks
|
|
2339
|
+
*/
|
|
2340
|
+
async listBenchmarks() {
|
|
2341
|
+
const res = await fetch(`${this.baseUrl}/api/storage/benchmarks`);
|
|
2342
|
+
if (!res.ok) {
|
|
2343
|
+
throw new Error(`Failed to list benchmarks: ${res.status} ${res.statusText}`);
|
|
2344
|
+
}
|
|
2345
|
+
const data = await res.json();
|
|
2346
|
+
return data.benchmarks || [];
|
|
2347
|
+
}
|
|
2348
|
+
/**
|
|
2349
|
+
* Get benchmark by ID
|
|
2350
|
+
*/
|
|
2351
|
+
async getBenchmark(id) {
|
|
2352
|
+
const res = await fetch(`${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(id)}`);
|
|
2353
|
+
if (res.status === 404) {
|
|
2354
|
+
return null;
|
|
2355
|
+
}
|
|
2356
|
+
if (!res.ok) {
|
|
2357
|
+
throw new Error(`Failed to get benchmark: ${res.status} ${res.statusText}`);
|
|
2358
|
+
}
|
|
2359
|
+
return res.json();
|
|
2360
|
+
}
|
|
2361
|
+
/**
|
|
2362
|
+
* Find benchmark by name or ID
|
|
2363
|
+
*
|
|
2364
|
+
* Prioritizes:
|
|
2365
|
+
* 1. Exact ID match
|
|
2366
|
+
* 2. Exact name match (case-sensitive)
|
|
2367
|
+
*/
|
|
2368
|
+
async findBenchmark(identifier) {
|
|
2369
|
+
const byId = await this.getBenchmark(identifier);
|
|
2370
|
+
if (byId) {
|
|
2371
|
+
return byId;
|
|
2372
|
+
}
|
|
2373
|
+
const benchmarks = await this.listBenchmarks();
|
|
2374
|
+
return benchmarks.find((b) => b.name === identifier) || null;
|
|
2375
|
+
}
|
|
2376
|
+
/**
|
|
2377
|
+
* Execute benchmark run (SSE stream)
|
|
2378
|
+
*
|
|
2379
|
+
* Streams progress events and returns the completed run.
|
|
2380
|
+
* If the SSE stream disconnects, falls back to polling for status.
|
|
2381
|
+
*/
|
|
2382
|
+
async executeBenchmark(benchmarkId, runConfig, onProgress) {
|
|
2383
|
+
const res = await fetch(
|
|
2384
|
+
`${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmarkId)}/execute`,
|
|
2385
|
+
{
|
|
2386
|
+
method: "POST",
|
|
2387
|
+
headers: { "Content-Type": "application/json" },
|
|
2388
|
+
body: JSON.stringify(runConfig)
|
|
2389
|
+
}
|
|
2390
|
+
);
|
|
2391
|
+
if (!res.ok) {
|
|
2392
|
+
const errorBody = await res.text();
|
|
2393
|
+
let errorMessage;
|
|
2394
|
+
try {
|
|
2395
|
+
const parsed = JSON.parse(errorBody);
|
|
2396
|
+
errorMessage = parsed.error || errorBody;
|
|
2397
|
+
} catch {
|
|
2398
|
+
errorMessage = errorBody;
|
|
2399
|
+
}
|
|
2400
|
+
throw new Error(`Failed to execute benchmark: ${errorMessage}`);
|
|
2401
|
+
}
|
|
2402
|
+
if (!res.body) {
|
|
2403
|
+
throw new Error("Response body is missing");
|
|
2404
|
+
}
|
|
2405
|
+
const reader = res.body.getReader();
|
|
2406
|
+
const decoder = new TextDecoder();
|
|
2407
|
+
let buffer = "";
|
|
2408
|
+
let finalRun = null;
|
|
2409
|
+
let runId = null;
|
|
2410
|
+
try {
|
|
2411
|
+
while (true) {
|
|
2412
|
+
const { done, value } = await reader.read();
|
|
2413
|
+
if (done) break;
|
|
2414
|
+
buffer += decoder.decode(value, { stream: true });
|
|
2415
|
+
const lines = buffer.split("\n\n");
|
|
2416
|
+
buffer = lines.pop() || "";
|
|
2417
|
+
for (const line of lines) {
|
|
2418
|
+
if (line.startsWith("data: ")) {
|
|
2419
|
+
try {
|
|
2420
|
+
const event = JSON.parse(line.slice(6));
|
|
2421
|
+
onProgress?.(event);
|
|
2422
|
+
if (event.type === "started") {
|
|
2423
|
+
runId = event.runId;
|
|
2424
|
+
} else if (event.type === "completed" || event.type === "cancelled") {
|
|
2425
|
+
finalRun = event.run;
|
|
2426
|
+
} else if (event.type === "error") {
|
|
2427
|
+
throw new Error(event.error);
|
|
2428
|
+
}
|
|
2429
|
+
} catch (e) {
|
|
2430
|
+
if (e instanceof SyntaxError) continue;
|
|
2431
|
+
throw e;
|
|
2432
|
+
}
|
|
2433
|
+
}
|
|
2434
|
+
}
|
|
2435
|
+
}
|
|
2436
|
+
} catch (streamError) {
|
|
2437
|
+
if (runId) {
|
|
2438
|
+
console.warn(`[ApiClient] SSE stream disconnected: ${streamError instanceof Error ? streamError.message : streamError}`);
|
|
2439
|
+
console.warn(`[ApiClient] Falling back to polling for run ${runId}...`);
|
|
2440
|
+
await new Promise((resolve5) => setTimeout(resolve5, 2e3));
|
|
2441
|
+
const polledRun = await this.pollRunStatus(benchmarkId, runId, (run) => {
|
|
2442
|
+
const completedCount = Object.values(run.results || {}).filter(
|
|
2443
|
+
(r) => r.status === "completed" || r.status === "failed"
|
|
2444
|
+
).length;
|
|
2445
|
+
const totalCount = Object.keys(run.results || {}).length;
|
|
2446
|
+
onProgress?.({
|
|
2447
|
+
type: "progress",
|
|
2448
|
+
currentTestCaseIndex: completedCount - 1,
|
|
2449
|
+
totalTestCases: totalCount,
|
|
2450
|
+
currentTestCase: { id: "polling", name: "Polling for status..." }
|
|
2451
|
+
});
|
|
2452
|
+
});
|
|
2453
|
+
if (polledRun) {
|
|
2454
|
+
return polledRun;
|
|
2455
|
+
}
|
|
2456
|
+
}
|
|
2457
|
+
throw streamError;
|
|
2458
|
+
} finally {
|
|
2459
|
+
try {
|
|
2460
|
+
await reader.cancel();
|
|
2461
|
+
} catch {
|
|
2462
|
+
}
|
|
2463
|
+
}
|
|
2464
|
+
if (!finalRun) {
|
|
2465
|
+
if (runId) {
|
|
2466
|
+
console.warn("[ApiClient] SSE stream ended without completion event, polling for status...");
|
|
2467
|
+
const polledRun = await this.pollRunStatus(benchmarkId, runId);
|
|
2468
|
+
if (polledRun) {
|
|
2469
|
+
return polledRun;
|
|
2470
|
+
}
|
|
2471
|
+
}
|
|
2472
|
+
throw new Error("No final run received from server");
|
|
2473
|
+
}
|
|
2474
|
+
return finalRun;
|
|
2475
|
+
}
|
|
2476
|
+
/**
|
|
2477
|
+
* Cancel an in-progress benchmark run
|
|
2478
|
+
*/
|
|
2479
|
+
async cancelRun(benchmarkId, runId) {
|
|
2480
|
+
const res = await fetch(
|
|
2481
|
+
`${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmarkId)}/cancel`,
|
|
2482
|
+
{
|
|
2483
|
+
method: "POST",
|
|
2484
|
+
headers: { "Content-Type": "application/json" },
|
|
2485
|
+
body: JSON.stringify({ runId })
|
|
2486
|
+
}
|
|
2487
|
+
);
|
|
2488
|
+
if (!res.ok) {
|
|
2489
|
+
const errorBody = await res.text();
|
|
2490
|
+
throw new Error(`Failed to cancel run: ${errorBody}`);
|
|
2491
|
+
}
|
|
2492
|
+
}
|
|
2493
|
+
/**
|
|
2494
|
+
* Get a specific run from a benchmark by ID.
|
|
2495
|
+
*
|
|
2496
|
+
* Fetches the benchmark and extracts the run with the matching ID.
|
|
2497
|
+
*
|
|
2498
|
+
* @param benchmarkId - The benchmark ID
|
|
2499
|
+
* @param runId - The run ID within the benchmark
|
|
2500
|
+
* @returns The run if found, null otherwise
|
|
2501
|
+
*/
|
|
2502
|
+
async getRun(benchmarkId, runId) {
|
|
2503
|
+
const benchmark = await this.getBenchmark(benchmarkId);
|
|
2504
|
+
if (!benchmark) {
|
|
2505
|
+
return null;
|
|
2506
|
+
}
|
|
2507
|
+
return benchmark.runs?.find((r) => r.id === runId) || null;
|
|
2508
|
+
}
|
|
2509
|
+
/**
|
|
2510
|
+
* Poll a run until it reaches a terminal state (completed, failed, cancelled).
|
|
2511
|
+
*
|
|
2512
|
+
* Used as a fallback when SSE stream connection is lost but server continues
|
|
2513
|
+
* processing in the background.
|
|
2514
|
+
*
|
|
2515
|
+
* @param benchmarkId - The benchmark ID
|
|
2516
|
+
* @param runId - The run ID to poll
|
|
2517
|
+
* @param onProgress - Optional callback for progress updates during polling
|
|
2518
|
+
* @param timeoutMs - Maximum time to wait (default: 10 minutes)
|
|
2519
|
+
* @returns The final run state, or null if not found
|
|
2520
|
+
*/
|
|
2521
|
+
async pollRunStatus(benchmarkId, runId, onProgress, timeoutMs = 6e5) {
|
|
2522
|
+
const startTime = Date.now();
|
|
2523
|
+
const pollInterval = 5e3;
|
|
2524
|
+
while (Date.now() - startTime < timeoutMs) {
|
|
2525
|
+
const run = await this.getRun(benchmarkId, runId);
|
|
2526
|
+
if (!run) return null;
|
|
2527
|
+
onProgress?.(run);
|
|
2528
|
+
if (run.status && ["completed", "failed", "cancelled"].includes(run.status)) {
|
|
2529
|
+
return run;
|
|
2530
|
+
}
|
|
2531
|
+
await new Promise((resolve5) => setTimeout(resolve5, pollInterval));
|
|
2532
|
+
}
|
|
2533
|
+
return this.getRun(benchmarkId, runId);
|
|
2534
|
+
}
|
|
2535
|
+
/**
|
|
2536
|
+
* Get a single report (TestCaseRun) by ID.
|
|
2537
|
+
*
|
|
2538
|
+
* This is the preferred method for fetching reports - use report IDs
|
|
2539
|
+
* from run.results[testCaseId].reportId.
|
|
2540
|
+
*
|
|
2541
|
+
* @param reportId - The report ID to fetch
|
|
2542
|
+
* @returns The report if found, null otherwise
|
|
2543
|
+
*/
|
|
2544
|
+
async getReportById(reportId) {
|
|
2545
|
+
const res = await fetch(
|
|
2546
|
+
`${this.baseUrl}/api/storage/runs/${encodeURIComponent(reportId)}`
|
|
2547
|
+
);
|
|
2548
|
+
if (res.status === 404) {
|
|
2549
|
+
return null;
|
|
2550
|
+
}
|
|
2551
|
+
if (!res.ok) {
|
|
2552
|
+
throw new Error(`Failed to get report: ${res.status} ${res.statusText}`);
|
|
2553
|
+
}
|
|
2554
|
+
return res.json();
|
|
2555
|
+
}
|
|
2556
|
+
/**
|
|
2557
|
+
* List all test cases (basic)
|
|
2558
|
+
*/
|
|
2559
|
+
async listTestCases() {
|
|
2560
|
+
const res = await fetch(`${this.baseUrl}/api/storage/test-cases`);
|
|
2561
|
+
if (!res.ok) {
|
|
2562
|
+
throw new Error(`Failed to list test cases: ${res.status} ${res.statusText}`);
|
|
2563
|
+
}
|
|
2564
|
+
const data = await res.json();
|
|
2565
|
+
return data.testCases || [];
|
|
2566
|
+
}
|
|
2567
|
+
/**
|
|
2568
|
+
* List all test cases with full data and metadata
|
|
2569
|
+
*/
|
|
2570
|
+
async listTestCasesWithMeta() {
|
|
2571
|
+
const res = await fetch(`${this.baseUrl}/api/storage/test-cases`);
|
|
2572
|
+
if (!res.ok) {
|
|
2573
|
+
throw new Error(`Failed to list test cases: ${res.status} ${res.statusText}`);
|
|
2574
|
+
}
|
|
2575
|
+
const data = await res.json();
|
|
2576
|
+
return {
|
|
2577
|
+
data: data.testCases || [],
|
|
2578
|
+
total: data.total || 0,
|
|
2579
|
+
meta: data.meta || {
|
|
2580
|
+
storageConfigured: false,
|
|
2581
|
+
storageReachable: false,
|
|
2582
|
+
realDataCount: 0,
|
|
2583
|
+
sampleDataCount: data.testCases?.length || 0
|
|
2584
|
+
}
|
|
2585
|
+
};
|
|
2586
|
+
}
|
|
2587
|
+
/**
|
|
2588
|
+
* Bulk create test cases
|
|
2589
|
+
*/
|
|
2590
|
+
async bulkCreateTestCases(testCases) {
|
|
2591
|
+
const res = await fetch(`${this.baseUrl}/api/storage/test-cases/bulk`, {
|
|
2592
|
+
method: "POST",
|
|
2593
|
+
headers: { "Content-Type": "application/json" },
|
|
2594
|
+
body: JSON.stringify({ testCases })
|
|
2595
|
+
});
|
|
2596
|
+
if (!res.ok) {
|
|
2597
|
+
const errorBody = await res.text();
|
|
2598
|
+
let errorMessage;
|
|
2599
|
+
try {
|
|
2600
|
+
const parsed = JSON.parse(errorBody);
|
|
2601
|
+
errorMessage = parsed.error || errorBody;
|
|
2602
|
+
} catch {
|
|
2603
|
+
errorMessage = errorBody;
|
|
2604
|
+
}
|
|
2605
|
+
throw new Error(`Failed to bulk create test cases: ${errorMessage}`);
|
|
2606
|
+
}
|
|
2607
|
+
return res.json();
|
|
2608
|
+
}
|
|
2609
|
+
/**
|
|
2610
|
+
* List all benchmarks with metadata
|
|
2611
|
+
*/
|
|
2612
|
+
async listBenchmarksWithMeta() {
|
|
2613
|
+
const res = await fetch(`${this.baseUrl}/api/storage/benchmarks`);
|
|
2614
|
+
if (!res.ok) {
|
|
2615
|
+
throw new Error(`Failed to list benchmarks: ${res.status} ${res.statusText}`);
|
|
2616
|
+
}
|
|
2617
|
+
const data = await res.json();
|
|
2618
|
+
return {
|
|
2619
|
+
data: data.benchmarks || [],
|
|
2620
|
+
total: data.total || 0,
|
|
2621
|
+
meta: data.meta || {
|
|
2622
|
+
storageConfigured: false,
|
|
2623
|
+
storageReachable: false,
|
|
2624
|
+
realDataCount: 0,
|
|
2625
|
+
sampleDataCount: data.benchmarks?.length || 0
|
|
2626
|
+
}
|
|
2627
|
+
};
|
|
2628
|
+
}
|
|
2629
|
+
/**
|
|
2630
|
+
* List all configured agents
|
|
2631
|
+
*/
|
|
2632
|
+
async listAgents() {
|
|
2633
|
+
const res = await fetch(`${this.baseUrl}/api/agents`);
|
|
2634
|
+
if (!res.ok) {
|
|
2635
|
+
throw new Error(`Failed to list agents: ${res.status} ${res.statusText}`);
|
|
2636
|
+
}
|
|
2637
|
+
const data = await res.json();
|
|
2638
|
+
return data.agents || [];
|
|
2639
|
+
}
|
|
2640
|
+
/**
|
|
2641
|
+
* List all configured models
|
|
2642
|
+
*/
|
|
2643
|
+
async listModels() {
|
|
2644
|
+
const res = await fetch(`${this.baseUrl}/api/models`);
|
|
2645
|
+
if (!res.ok) {
|
|
2646
|
+
throw new Error(`Failed to list models: ${res.status} ${res.statusText}`);
|
|
2647
|
+
}
|
|
2648
|
+
const data = await res.json();
|
|
2649
|
+
return data.models || [];
|
|
2650
|
+
}
|
|
2651
|
+
/**
|
|
2652
|
+
* Get a single test case by ID
|
|
2653
|
+
*/
|
|
2654
|
+
async getTestCase(id) {
|
|
2655
|
+
const res = await fetch(`${this.baseUrl}/api/storage/test-cases/${encodeURIComponent(id)}`);
|
|
2656
|
+
if (res.status === 404) {
|
|
2657
|
+
return null;
|
|
2658
|
+
}
|
|
2659
|
+
if (!res.ok) {
|
|
2660
|
+
throw new Error(`Failed to get test case: ${res.status} ${res.statusText}`);
|
|
2661
|
+
}
|
|
2662
|
+
return res.json();
|
|
2663
|
+
}
|
|
2664
|
+
/**
|
|
2665
|
+
* Find test case by ID or name
|
|
2666
|
+
*/
|
|
2667
|
+
async findTestCase(identifier) {
|
|
2668
|
+
const byId = await this.getTestCase(identifier);
|
|
2669
|
+
if (byId) {
|
|
2670
|
+
return byId;
|
|
2671
|
+
}
|
|
2672
|
+
const response = await this.listTestCasesWithMeta();
|
|
2673
|
+
return response.data.find(
|
|
2674
|
+
(tc) => tc.name.toLowerCase() === identifier.toLowerCase()
|
|
2675
|
+
) || null;
|
|
2676
|
+
}
|
|
2677
|
+
/**
|
|
2678
|
+
* Evaluation progress event types
|
|
2679
|
+
*/
|
|
2680
|
+
/**
|
|
2681
|
+
* Run a single test case evaluation via server API (SSE stream)
|
|
2682
|
+
*/
|
|
2683
|
+
async runEvaluation(testCaseId, agentKey, modelId, onProgress) {
|
|
2684
|
+
const res = await fetch(`${this.baseUrl}/api/evaluate`, {
|
|
2685
|
+
method: "POST",
|
|
2686
|
+
headers: { "Content-Type": "application/json" },
|
|
2687
|
+
body: JSON.stringify({ testCaseId, agentKey, modelId })
|
|
2688
|
+
});
|
|
2689
|
+
if (!res.ok) {
|
|
2690
|
+
const errorBody = await res.text();
|
|
2691
|
+
let errorMessage;
|
|
2692
|
+
try {
|
|
2693
|
+
const parsed = JSON.parse(errorBody);
|
|
2694
|
+
errorMessage = parsed.error || errorBody;
|
|
2695
|
+
} catch {
|
|
2696
|
+
errorMessage = errorBody;
|
|
2697
|
+
}
|
|
2698
|
+
throw new Error(`Failed to run evaluation: ${errorMessage}`);
|
|
2699
|
+
}
|
|
2700
|
+
if (!res.body) {
|
|
2701
|
+
throw new Error("Response body is missing");
|
|
2702
|
+
}
|
|
2703
|
+
const reader = res.body.getReader();
|
|
2704
|
+
const decoder = new TextDecoder();
|
|
2705
|
+
let buffer = "";
|
|
2706
|
+
let result = null;
|
|
2707
|
+
try {
|
|
2708
|
+
while (true) {
|
|
2709
|
+
const { done, value } = await reader.read();
|
|
2710
|
+
if (done) break;
|
|
2711
|
+
buffer += decoder.decode(value, { stream: true });
|
|
2712
|
+
const lines = buffer.split("\n\n");
|
|
2713
|
+
buffer = lines.pop() || "";
|
|
2714
|
+
for (const line of lines) {
|
|
2715
|
+
if (line.startsWith("data: ")) {
|
|
2716
|
+
try {
|
|
2717
|
+
const event = JSON.parse(line.slice(6));
|
|
2718
|
+
onProgress?.(event);
|
|
2719
|
+
if (event.type === "completed") {
|
|
2720
|
+
result = event.report;
|
|
2721
|
+
} else if (event.type === "error") {
|
|
2722
|
+
throw new Error(event.error);
|
|
2723
|
+
}
|
|
2724
|
+
} catch (e) {
|
|
2725
|
+
if (e instanceof SyntaxError) continue;
|
|
2726
|
+
throw e;
|
|
2727
|
+
}
|
|
2728
|
+
}
|
|
2729
|
+
}
|
|
2730
|
+
}
|
|
2731
|
+
} finally {
|
|
2732
|
+
try {
|
|
2733
|
+
await reader.cancel();
|
|
2734
|
+
} catch {
|
|
2735
|
+
}
|
|
2736
|
+
}
|
|
2737
|
+
if (!result) {
|
|
2738
|
+
throw new Error("No result received from evaluation");
|
|
2739
|
+
}
|
|
2740
|
+
return result;
|
|
2741
|
+
}
|
|
2742
|
+
/**
|
|
2743
|
+
* Export benchmark test cases as import-compatible JSON
|
|
2744
|
+
*/
|
|
2745
|
+
async exportBenchmark(benchmarkId) {
|
|
2746
|
+
const res = await fetch(
|
|
2747
|
+
`${this.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmarkId)}/export`
|
|
2748
|
+
);
|
|
2749
|
+
if (res.status === 404) {
|
|
2750
|
+
throw new Error(`Benchmark not found: ${benchmarkId}`);
|
|
2751
|
+
}
|
|
2752
|
+
if (!res.ok) {
|
|
2753
|
+
throw new Error(`Failed to export benchmark: ${res.status} ${res.statusText}`);
|
|
2754
|
+
}
|
|
2755
|
+
return res.json();
|
|
2756
|
+
}
|
|
2757
|
+
/**
|
|
2758
|
+
* Create a new benchmark
|
|
2759
|
+
*/
|
|
2760
|
+
async createBenchmark(input) {
|
|
2761
|
+
const res = await fetch(`${this.baseUrl}/api/storage/benchmarks`, {
|
|
2762
|
+
method: "POST",
|
|
2763
|
+
headers: { "Content-Type": "application/json" },
|
|
2764
|
+
body: JSON.stringify(input)
|
|
2765
|
+
});
|
|
2766
|
+
if (!res.ok) {
|
|
2767
|
+
const errorBody = await res.text();
|
|
2768
|
+
throw new Error(`Failed to create benchmark: ${errorBody}`);
|
|
2769
|
+
}
|
|
2770
|
+
return res.json();
|
|
2771
|
+
}
|
|
2772
|
+
};
|
|
2773
|
+
|
|
2774
|
+
// cli/commands/list.ts
|
|
2775
|
+
function formatJson(data) {
|
|
2776
|
+
return JSON.stringify(data, null, 2);
|
|
2777
|
+
}
|
|
2778
|
+
function displayStorageWarnings(meta) {
|
|
2779
|
+
if (!meta.storageConfigured) {
|
|
2780
|
+
console.log(chalk.yellow("\n \u26A0 Storage not configured"));
|
|
2781
|
+
console.log(chalk.gray(" Showing sample data only. Set OPENSEARCH_STORAGE_* env vars for persistent storage.\n"));
|
|
2782
|
+
} else if (!meta.storageReachable) {
|
|
2783
|
+
console.log(chalk.yellow("\n \u26A0 Storage unreachable"));
|
|
2784
|
+
meta.warnings?.forEach((w) => console.log(chalk.gray(` - ${w}`)));
|
|
2785
|
+
console.log();
|
|
2786
|
+
}
|
|
2787
|
+
}
|
|
2788
|
+
async function listAgents(format, config) {
|
|
2789
|
+
const serverResult = await ensureServer(config.server);
|
|
2790
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
2791
|
+
try {
|
|
2792
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
2793
|
+
const agents = await client.listAgents();
|
|
2794
|
+
if (format === "json") {
|
|
2795
|
+
console.log(formatJson(agents));
|
|
2796
|
+
return;
|
|
2797
|
+
}
|
|
2798
|
+
const table = new Table({
|
|
2799
|
+
head: [
|
|
2800
|
+
chalk.cyan("Key"),
|
|
2801
|
+
chalk.cyan("Name"),
|
|
2802
|
+
chalk.cyan("Connector"),
|
|
2803
|
+
chalk.cyan("Models"),
|
|
2804
|
+
chalk.cyan("Endpoint")
|
|
2805
|
+
],
|
|
2806
|
+
colWidths: [15, 20, 15, 25, 40],
|
|
2807
|
+
wordWrap: true
|
|
2808
|
+
});
|
|
2809
|
+
for (const agent of agents) {
|
|
2810
|
+
table.push([
|
|
2811
|
+
agent.key,
|
|
2812
|
+
agent.name,
|
|
2813
|
+
agent.connectorType || "agui-streaming",
|
|
2814
|
+
agent.models.slice(0, 3).join(", ") + (agent.models.length > 3 ? "..." : ""),
|
|
2815
|
+
agent.endpoint.substring(0, 37) + (agent.endpoint.length > 37 ? "..." : "")
|
|
2816
|
+
]);
|
|
2817
|
+
}
|
|
2818
|
+
console.log(chalk.bold("\nAvailable Agents:\n"));
|
|
2819
|
+
console.log(table.toString());
|
|
2820
|
+
console.log(chalk.gray(`
|
|
2821
|
+
Total: ${agents.length} agents
|
|
2822
|
+
`));
|
|
2823
|
+
} catch (error) {
|
|
2824
|
+
console.error(chalk.red(`
|
|
2825
|
+
Error: ${error.message}`));
|
|
2826
|
+
console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
|
|
2827
|
+
process.exit(1);
|
|
2828
|
+
} finally {
|
|
2829
|
+
cleanup();
|
|
2830
|
+
}
|
|
2831
|
+
}
|
|
2832
|
+
async function listTestCases(format, config) {
|
|
2833
|
+
const serverResult = await ensureServer(config.server);
|
|
2834
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
2835
|
+
try {
|
|
2836
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
2837
|
+
const response = await client.listTestCasesWithMeta();
|
|
2838
|
+
displayStorageWarnings(response.meta);
|
|
2839
|
+
if (format === "json") {
|
|
2840
|
+
console.log(formatJson(response));
|
|
2841
|
+
return;
|
|
2842
|
+
}
|
|
2843
|
+
const table = new Table({
|
|
2844
|
+
head: [
|
|
2845
|
+
chalk.cyan("ID"),
|
|
2846
|
+
chalk.cyan("Name"),
|
|
2847
|
+
chalk.cyan("Labels"),
|
|
2848
|
+
chalk.cyan("Version"),
|
|
2849
|
+
chalk.cyan("Source")
|
|
2850
|
+
],
|
|
2851
|
+
colWidths: [25, 28, 28, 10, 10],
|
|
2852
|
+
wordWrap: true
|
|
2853
|
+
});
|
|
2854
|
+
for (const tc of response.data) {
|
|
2855
|
+
const isDemo = tc.id.startsWith("demo-");
|
|
2856
|
+
table.push([
|
|
2857
|
+
tc.id,
|
|
2858
|
+
tc.name,
|
|
2859
|
+
tc.labels?.slice(0, 3).join(", ") || "",
|
|
2860
|
+
`v${tc.currentVersion || 1}`,
|
|
2861
|
+
isDemo ? chalk.gray("Sample") : chalk.green("Stored")
|
|
2862
|
+
]);
|
|
2863
|
+
}
|
|
2864
|
+
console.log(chalk.bold("\nAvailable Test Cases:\n"));
|
|
2865
|
+
console.log(table.toString());
|
|
2866
|
+
const { meta } = response;
|
|
2867
|
+
if (meta.realDataCount > 0 || meta.sampleDataCount > 0) {
|
|
2868
|
+
console.log(chalk.gray(`
|
|
2869
|
+
Total: ${response.total} test cases (${meta.realDataCount} stored, ${meta.sampleDataCount} sample)
|
|
2870
|
+
`));
|
|
2871
|
+
} else {
|
|
2872
|
+
console.log(chalk.gray(`
|
|
2873
|
+
Total: ${response.total} test cases
|
|
2874
|
+
`));
|
|
2875
|
+
}
|
|
2876
|
+
} catch (error) {
|
|
2877
|
+
console.error(chalk.red(`
|
|
2878
|
+
Error: ${error.message}`));
|
|
2879
|
+
console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
|
|
2880
|
+
process.exit(1);
|
|
2881
|
+
} finally {
|
|
2882
|
+
cleanup();
|
|
2883
|
+
}
|
|
2884
|
+
}
|
|
2885
|
+
async function listBenchmarks(format, config) {
|
|
2886
|
+
const serverResult = await ensureServer(config.server);
|
|
2887
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
2888
|
+
try {
|
|
2889
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
2890
|
+
const response = await client.listBenchmarksWithMeta();
|
|
2891
|
+
displayStorageWarnings(response.meta);
|
|
2892
|
+
if (format === "json") {
|
|
2893
|
+
console.log(formatJson(response));
|
|
2894
|
+
return;
|
|
2895
|
+
}
|
|
2896
|
+
const table = new Table({
|
|
2897
|
+
head: [
|
|
2898
|
+
chalk.cyan("ID"),
|
|
2899
|
+
chalk.cyan("Name"),
|
|
2900
|
+
chalk.cyan("Test Cases"),
|
|
2901
|
+
chalk.cyan("Created"),
|
|
2902
|
+
chalk.cyan("Source")
|
|
2903
|
+
],
|
|
2904
|
+
colWidths: [28, 28, 12, 22, 10],
|
|
2905
|
+
wordWrap: true
|
|
2906
|
+
});
|
|
2907
|
+
for (const b of response.data) {
|
|
2908
|
+
const isDemo = b.id.startsWith("demo-");
|
|
2909
|
+
table.push([
|
|
2910
|
+
b.id,
|
|
2911
|
+
b.name,
|
|
2912
|
+
b.testCaseIds.length.toString(),
|
|
2913
|
+
new Date(b.createdAt).toLocaleDateString(),
|
|
2914
|
+
isDemo ? chalk.gray("Sample") : chalk.green("Stored")
|
|
2915
|
+
]);
|
|
2916
|
+
}
|
|
2917
|
+
console.log(chalk.bold("\nAvailable Benchmarks:\n"));
|
|
2918
|
+
console.log(table.toString());
|
|
2919
|
+
const { meta } = response;
|
|
2920
|
+
if (meta.realDataCount > 0 || meta.sampleDataCount > 0) {
|
|
2921
|
+
console.log(chalk.gray(`
|
|
2922
|
+
Total: ${response.total} benchmarks (${meta.realDataCount} stored, ${meta.sampleDataCount} sample)
|
|
2923
|
+
`));
|
|
2924
|
+
} else {
|
|
2925
|
+
console.log(chalk.gray(`
|
|
2926
|
+
Total: ${response.total} benchmarks
|
|
2927
|
+
`));
|
|
2928
|
+
}
|
|
2929
|
+
} catch (error) {
|
|
2930
|
+
console.error(chalk.red(`
|
|
2931
|
+
Error: ${error.message}`));
|
|
2932
|
+
console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
|
|
2933
|
+
process.exit(1);
|
|
2934
|
+
} finally {
|
|
2935
|
+
cleanup();
|
|
2936
|
+
}
|
|
2937
|
+
}
|
|
2938
|
+
function listConnectors(format) {
|
|
2939
|
+
const types = connectorRegistry.getRegisteredTypes();
|
|
2940
|
+
const connectors = types.map((type) => {
|
|
2941
|
+
const connector = connectorRegistry.get(type);
|
|
2942
|
+
return {
|
|
2943
|
+
type,
|
|
2944
|
+
name: connector?.name || "Unknown",
|
|
2945
|
+
streaming: connector?.supportsStreaming || false
|
|
2946
|
+
};
|
|
2947
|
+
});
|
|
2948
|
+
if (format === "json") {
|
|
2949
|
+
console.log(formatJson(connectors));
|
|
2950
|
+
return;
|
|
2951
|
+
}
|
|
2952
|
+
const table = new Table({
|
|
2953
|
+
head: [
|
|
2954
|
+
chalk.cyan("Type"),
|
|
2955
|
+
chalk.cyan("Name"),
|
|
2956
|
+
chalk.cyan("Streaming")
|
|
2957
|
+
],
|
|
2958
|
+
colWidths: [20, 25, 12]
|
|
2959
|
+
});
|
|
2960
|
+
for (const c of connectors) {
|
|
2961
|
+
table.push([
|
|
2962
|
+
c.type,
|
|
2963
|
+
c.name,
|
|
2964
|
+
c.streaming ? chalk.green("Yes") : chalk.gray("No")
|
|
2965
|
+
]);
|
|
2966
|
+
}
|
|
2967
|
+
console.log(chalk.bold("\nRegistered Connectors:\n"));
|
|
2968
|
+
console.log(table.toString());
|
|
2969
|
+
console.log(chalk.gray(`
|
|
2970
|
+
Total: ${connectors.length} connectors
|
|
2971
|
+
`));
|
|
2972
|
+
}
|
|
2973
|
+
async function listModels(format, config) {
|
|
2974
|
+
const serverResult = await ensureServer(config.server);
|
|
2975
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
2976
|
+
try {
|
|
2977
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
2978
|
+
const models = await client.listModels();
|
|
2979
|
+
if (format === "json") {
|
|
2980
|
+
console.log(formatJson(models));
|
|
2981
|
+
return;
|
|
2982
|
+
}
|
|
2983
|
+
const table = new Table({
|
|
2984
|
+
head: [
|
|
2985
|
+
chalk.cyan("Key"),
|
|
2986
|
+
chalk.cyan("Display Name"),
|
|
2987
|
+
chalk.cyan("Provider"),
|
|
2988
|
+
chalk.cyan("Context")
|
|
2989
|
+
],
|
|
2990
|
+
colWidths: [25, 30, 12, 12],
|
|
2991
|
+
wordWrap: true
|
|
2992
|
+
});
|
|
2993
|
+
for (const m of models) {
|
|
2994
|
+
table.push([
|
|
2995
|
+
m.key,
|
|
2996
|
+
m.display_name || m.key,
|
|
2997
|
+
m.provider || "bedrock",
|
|
2998
|
+
m.context_window ? `${Math.round(m.context_window / 1e3)}k` : "-"
|
|
2999
|
+
]);
|
|
3000
|
+
}
|
|
3001
|
+
console.log(chalk.bold("\nAvailable Models:\n"));
|
|
3002
|
+
console.log(table.toString());
|
|
3003
|
+
console.log(chalk.gray(`
|
|
3004
|
+
Total: ${models.length} models
|
|
3005
|
+
`));
|
|
3006
|
+
} catch (error) {
|
|
3007
|
+
console.error(chalk.red(`
|
|
3008
|
+
Error: ${error.message}`));
|
|
3009
|
+
console.log(chalk.gray(" Is the server running? Start with: npm run dev:server\n"));
|
|
3010
|
+
process.exit(1);
|
|
3011
|
+
} finally {
|
|
3012
|
+
cleanup();
|
|
3013
|
+
}
|
|
3014
|
+
}
|
|
3015
|
+
function createListCommand() {
|
|
3016
|
+
const command = new Command("list").description("List available resources").argument("<resource>", "Resource type: agents, test-cases, benchmarks, connectors, models").option("-o, --output <format>", "Output format: table, json", "table").action(async (resource, options) => {
|
|
3017
|
+
const format = options.output;
|
|
3018
|
+
const config = await loadConfig();
|
|
3019
|
+
for (const connector of config.connectors) {
|
|
3020
|
+
connectorRegistry.register(connector);
|
|
3021
|
+
}
|
|
3022
|
+
switch (resource.toLowerCase()) {
|
|
3023
|
+
case "agents":
|
|
3024
|
+
await listAgents(format, config);
|
|
3025
|
+
break;
|
|
3026
|
+
case "test-cases":
|
|
3027
|
+
case "testcases":
|
|
3028
|
+
case "tc":
|
|
3029
|
+
await listTestCases(format, config);
|
|
3030
|
+
break;
|
|
3031
|
+
case "benchmarks":
|
|
3032
|
+
case "bench":
|
|
3033
|
+
await listBenchmarks(format, config);
|
|
3034
|
+
break;
|
|
3035
|
+
case "connectors":
|
|
3036
|
+
listConnectors(format);
|
|
3037
|
+
break;
|
|
3038
|
+
case "models":
|
|
3039
|
+
await listModels(format, config);
|
|
3040
|
+
break;
|
|
3041
|
+
default:
|
|
3042
|
+
console.error(chalk.red(`
|
|
3043
|
+
Unknown resource type: ${resource}`));
|
|
3044
|
+
console.log(chalk.gray(" Available: agents, test-cases, benchmarks, connectors, models\n"));
|
|
3045
|
+
process.exit(1);
|
|
3046
|
+
}
|
|
3047
|
+
});
|
|
3048
|
+
return command;
|
|
3049
|
+
}
|
|
3050
|
+
|
|
3051
|
+
// cli/commands/run.ts
|
|
3052
|
+
import { Command as Command2 } from "commander";
|
|
3053
|
+
import chalk2 from "chalk";
|
|
3054
|
+
import ora from "ora";
|
|
3055
|
+
import Table2 from "cli-table3";
|
|
3056
|
+
function findAgent(identifier, config) {
|
|
3057
|
+
return config.agents.find(
|
|
3058
|
+
(a) => a.key === identifier || a.name.toLowerCase() === identifier.toLowerCase()
|
|
3059
|
+
);
|
|
3060
|
+
}
|
|
3061
|
+
function getDefaultModel(agent) {
|
|
3062
|
+
return agent.models[0] || "claude-sonnet";
|
|
3063
|
+
}
|
|
3064
|
+
async function commandExists(command) {
|
|
3065
|
+
const { execSync: execSync2 } = await import("child_process");
|
|
3066
|
+
const checkCommand = process.platform === "win32" ? `where ${command}` : `which ${command}`;
|
|
3067
|
+
try {
|
|
3068
|
+
execSync2(checkCommand, { stdio: "ignore" });
|
|
3069
|
+
return true;
|
|
3070
|
+
} catch {
|
|
3071
|
+
return false;
|
|
3072
|
+
}
|
|
3073
|
+
}
|
|
3074
|
+
async function validateAgentRequirements(agent) {
|
|
3075
|
+
if (agent.connectorType === "claude-code") {
|
|
3076
|
+
if (!await commandExists("claude")) {
|
|
3077
|
+
return `Claude CLI not found in PATH. Install Claude Code CLI: https://claude.ai/code`;
|
|
3078
|
+
}
|
|
3079
|
+
}
|
|
3080
|
+
if (agent.connectorType === "subprocess" && agent.endpoint) {
|
|
3081
|
+
const command = agent.endpoint.split(" ")[0];
|
|
3082
|
+
if (!await commandExists(command)) {
|
|
3083
|
+
return `Command '${command}' not found in PATH`;
|
|
3084
|
+
}
|
|
3085
|
+
}
|
|
3086
|
+
return void 0;
|
|
3087
|
+
}
|
|
3088
|
+
async function runForAgent(client, testCaseId, agent, modelId, verbose) {
|
|
3089
|
+
const spinner = ora(`Running ${agent.name}...`).start();
|
|
3090
|
+
try {
|
|
3091
|
+
const report = await client.runEvaluation(
|
|
3092
|
+
testCaseId,
|
|
3093
|
+
agent.key,
|
|
3094
|
+
modelId,
|
|
3095
|
+
(event) => {
|
|
3096
|
+
if (event.type === "step" && verbose) {
|
|
3097
|
+
spinner.text = `${agent.name}: Step ${event.stepIndex + 1} (${event.step.type})`;
|
|
3098
|
+
} else if (event.type === "started") {
|
|
3099
|
+
spinner.text = `${agent.name}: Started evaluation...`;
|
|
3100
|
+
}
|
|
3101
|
+
}
|
|
3102
|
+
);
|
|
3103
|
+
if (report.status === "completed" && report.passFailStatus === "passed") {
|
|
3104
|
+
spinner.succeed(`${agent.name}: ${chalk2.green("PASSED")}`);
|
|
3105
|
+
} else if (report.status === "completed") {
|
|
3106
|
+
spinner.succeed(`${agent.name}: ${chalk2.red("FAILED")}`);
|
|
3107
|
+
} else {
|
|
3108
|
+
spinner.fail(`${agent.name}: ${chalk2.yellow(report.status)}`);
|
|
3109
|
+
}
|
|
3110
|
+
return report;
|
|
3111
|
+
} catch (error) {
|
|
3112
|
+
spinner.fail(`${agent.name}: ${chalk2.red("ERROR")}`);
|
|
3113
|
+
throw error;
|
|
3114
|
+
}
|
|
3115
|
+
}
|
|
3116
|
+
function displayTableResults(results) {
|
|
3117
|
+
const table = new Table2({
|
|
3118
|
+
head: [
|
|
3119
|
+
chalk2.cyan("Agent"),
|
|
3120
|
+
chalk2.cyan("Status"),
|
|
3121
|
+
chalk2.cyan("Accuracy"),
|
|
3122
|
+
chalk2.cyan("Steps"),
|
|
3123
|
+
chalk2.cyan("Report ID")
|
|
3124
|
+
],
|
|
3125
|
+
colWidths: [20, 12, 12, 10, 30]
|
|
3126
|
+
});
|
|
3127
|
+
for (const r of results) {
|
|
3128
|
+
if (!r.report) {
|
|
3129
|
+
table.push([
|
|
3130
|
+
r.agent.name,
|
|
3131
|
+
chalk2.red("ERROR"),
|
|
3132
|
+
"-",
|
|
3133
|
+
"-",
|
|
3134
|
+
"-"
|
|
3135
|
+
]);
|
|
3136
|
+
continue;
|
|
3137
|
+
}
|
|
3138
|
+
const statusStr = r.report.passFailStatus === "passed" ? chalk2.green("PASSED") : r.report.passFailStatus === "failed" ? chalk2.red("FAILED") : chalk2.yellow(r.report.status);
|
|
3139
|
+
table.push([
|
|
3140
|
+
r.agent.name,
|
|
3141
|
+
statusStr,
|
|
3142
|
+
r.report.metrics?.accuracy ? `${Math.round(r.report.metrics.accuracy)}%` : "-",
|
|
3143
|
+
r.report.trajectorySteps.toString(),
|
|
3144
|
+
r.report.id?.substring(0, 27) + "..." || "-"
|
|
3145
|
+
]);
|
|
3146
|
+
}
|
|
3147
|
+
console.log("\n");
|
|
3148
|
+
console.log(table.toString());
|
|
3149
|
+
}
|
|
3150
|
+
function createRunCommand() {
|
|
3151
|
+
const command = new Command2("run").description("Run a test case against agents").requiredOption("-t, --test-case <id>", "Test case ID or name").option("-a, --agent <key>", "Agent key (can be specified multiple times)", (val, arr) => [...arr, val], []).option("-m, --model <id>", "Model ID (uses agent default if not specified)").option("-o, --output <format>", "Output format: table, json", "table").option("-v, --verbose", "Show detailed trajectory output").action(async (options) => {
|
|
3152
|
+
console.log(chalk2.bold("\nAgent Health - Test Case Runner\n"));
|
|
3153
|
+
const config = await loadConfig();
|
|
3154
|
+
for (const connector of config.connectors) {
|
|
3155
|
+
connectorRegistry.register(connector);
|
|
3156
|
+
}
|
|
3157
|
+
const serverResult = await ensureServer(config.server);
|
|
3158
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
3159
|
+
try {
|
|
3160
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
3161
|
+
await client.checkHealth();
|
|
3162
|
+
const testCase = await client.findTestCase(options.testCase);
|
|
3163
|
+
if (!testCase) {
|
|
3164
|
+
console.error(chalk2.red(` Error: Test case not found: ${options.testCase}`));
|
|
3165
|
+
console.log(chalk2.gray(" Use `agent-health list test-cases` to see available test cases\n"));
|
|
3166
|
+
process.exit(1);
|
|
3167
|
+
}
|
|
3168
|
+
console.log(chalk2.gray(` Test Case: ${testCase.name} (${testCase.id})`));
|
|
3169
|
+
console.log(chalk2.gray(` Server: ${serverResult.baseUrl}`));
|
|
3170
|
+
let agents = [];
|
|
3171
|
+
if (options.agent.length === 0) {
|
|
3172
|
+
agents = [config.agents[0]];
|
|
3173
|
+
console.log(chalk2.gray(` Agent: ${agents[0].name} (default)`));
|
|
3174
|
+
} else {
|
|
3175
|
+
for (const agentId of options.agent) {
|
|
3176
|
+
const agent = findAgent(agentId, config);
|
|
3177
|
+
if (!agent) {
|
|
3178
|
+
console.error(chalk2.red(` Error: Agent not found: ${agentId}`));
|
|
3179
|
+
console.log(chalk2.gray(" Use `agent-health list agents` to see available agents\n"));
|
|
3180
|
+
process.exit(1);
|
|
3181
|
+
}
|
|
3182
|
+
agents.push(agent);
|
|
3183
|
+
}
|
|
3184
|
+
console.log(chalk2.gray(` Agents: ${agents.map((a) => a.name).join(", ")}`));
|
|
3185
|
+
}
|
|
3186
|
+
console.log("");
|
|
3187
|
+
const results = [];
|
|
3188
|
+
for (const agent of agents) {
|
|
3189
|
+
const modelId = options.model || getDefaultModel(agent);
|
|
3190
|
+
const validationError = await validateAgentRequirements(agent);
|
|
3191
|
+
if (validationError) {
|
|
3192
|
+
console.error(chalk2.red(` Error: ${validationError}`));
|
|
3193
|
+
results.push({ agent, report: null });
|
|
3194
|
+
continue;
|
|
3195
|
+
}
|
|
3196
|
+
try {
|
|
3197
|
+
const report = await runForAgent(client, testCase.id, agent, modelId, options.verbose || false);
|
|
3198
|
+
results.push({ agent, report });
|
|
3199
|
+
} catch (error) {
|
|
3200
|
+
console.error(chalk2.red(` Error running ${agent.name}: ${error instanceof Error ? error.message : error}`));
|
|
3201
|
+
results.push({ agent, report: null });
|
|
3202
|
+
}
|
|
3203
|
+
}
|
|
3204
|
+
if (options.output === "json") {
|
|
3205
|
+
console.log(JSON.stringify(results.map((r) => ({
|
|
3206
|
+
agent: { key: r.agent.key, name: r.agent.name },
|
|
3207
|
+
report: r.report
|
|
3208
|
+
})), null, 2));
|
|
3209
|
+
} else {
|
|
3210
|
+
displayTableResults(results);
|
|
3211
|
+
}
|
|
3212
|
+
} catch (error) {
|
|
3213
|
+
console.error(chalk2.red(`
|
|
3214
|
+
Error: ${error.message}`));
|
|
3215
|
+
console.log(chalk2.gray(" Is the server running? Start with: npm run dev:server\n"));
|
|
3216
|
+
process.exit(1);
|
|
3217
|
+
} finally {
|
|
3218
|
+
cleanup();
|
|
3219
|
+
}
|
|
3220
|
+
});
|
|
3221
|
+
return command;
|
|
3222
|
+
}
|
|
3223
|
+
|
|
3224
|
+
// cli/commands/benchmark.ts
|
|
3225
|
+
import { Command as Command3 } from "commander";
|
|
3226
|
+
import chalk3 from "chalk";
|
|
3227
|
+
import ora2 from "ora";
|
|
3228
|
+
import Table3 from "cli-table3";
|
|
3229
|
+
import { readFileSync as readFileSync2, writeFileSync } from "fs";
|
|
3230
|
+
|
|
3231
|
+
// lib/testCaseValidation.ts
|
|
3232
|
+
import { z } from "zod";
|
|
3233
|
+
var contextItemSchema = z.object({
|
|
3234
|
+
description: z.string(),
|
|
3235
|
+
value: z.string()
|
|
3236
|
+
}).required();
|
|
3237
|
+
var difficultySchema = z.enum(["Easy", "Medium", "Hard"]);
|
|
3238
|
+
var testCaseSchema = z.object({
|
|
3239
|
+
name: z.string().min(1, "Name is required"),
|
|
3240
|
+
description: z.string().optional().default(""),
|
|
3241
|
+
category: z.string().min(1, "Category is required"),
|
|
3242
|
+
subcategory: z.string().optional(),
|
|
3243
|
+
difficulty: difficultySchema,
|
|
3244
|
+
initialPrompt: z.string().min(1, "Initial prompt is required"),
|
|
3245
|
+
context: z.array(contextItemSchema).optional().default([]),
|
|
3246
|
+
expectedOutcomes: z.array(z.string()).refine(
|
|
3247
|
+
(outcomes) => outcomes.some((o) => o.trim().length > 0),
|
|
3248
|
+
"At least one non-empty expected outcome is required"
|
|
3249
|
+
)
|
|
3250
|
+
});
|
|
3251
|
+
var testCasesArraySchema = z.array(testCaseSchema).min(1, "Array cannot be empty");
|
|
3252
|
+
function zodErrorToValidationErrors(error) {
|
|
3253
|
+
return error.errors.map((e) => ({
|
|
3254
|
+
path: e.path.join("."),
|
|
3255
|
+
message: e.message,
|
|
3256
|
+
type: "error"
|
|
3257
|
+
}));
|
|
3258
|
+
}
|
|
3259
|
+
function validateTestCaseJson(json) {
|
|
3260
|
+
if (Array.isArray(json)) {
|
|
3261
|
+
return {
|
|
3262
|
+
valid: false,
|
|
3263
|
+
errors: [{ path: "", message: "Received an array. Use Bulk Import mode for multiple test cases", type: "error" }]
|
|
3264
|
+
};
|
|
3265
|
+
}
|
|
3266
|
+
const result = testCaseSchema.safeParse(json);
|
|
3267
|
+
if (!result.success) {
|
|
3268
|
+
return {
|
|
3269
|
+
valid: false,
|
|
3270
|
+
errors: zodErrorToValidationErrors(result.error)
|
|
3271
|
+
};
|
|
3272
|
+
}
|
|
3273
|
+
return { valid: true, errors: [], data: result.data };
|
|
3274
|
+
}
|
|
3275
|
+
function validateTestCasesArrayJson(json) {
|
|
3276
|
+
if (json && typeof json === "object" && !Array.isArray(json)) {
|
|
3277
|
+
const singleResult = validateTestCaseJson(json);
|
|
3278
|
+
if (singleResult.valid && singleResult.data) {
|
|
3279
|
+
return {
|
|
3280
|
+
valid: true,
|
|
3281
|
+
errors: [],
|
|
3282
|
+
data: [singleResult.data]
|
|
3283
|
+
};
|
|
3284
|
+
}
|
|
3285
|
+
return { valid: false, errors: singleResult.errors };
|
|
3286
|
+
}
|
|
3287
|
+
const result = testCasesArraySchema.safeParse(json);
|
|
3288
|
+
if (!result.success) {
|
|
3289
|
+
return {
|
|
3290
|
+
valid: false,
|
|
3291
|
+
errors: zodErrorToValidationErrors(result.error)
|
|
3292
|
+
};
|
|
3293
|
+
}
|
|
3294
|
+
return { valid: true, errors: [], data: result.data };
|
|
3295
|
+
}
|
|
3296
|
+
|
|
3297
|
+
// lib/runStats.ts
|
|
3298
|
+
function calculateRunStats(run, reports) {
|
|
3299
|
+
let passed = 0;
|
|
3300
|
+
let failed = 0;
|
|
3301
|
+
let pending = 0;
|
|
3302
|
+
let total = 0;
|
|
3303
|
+
Object.entries(run.results || {}).forEach(([testCaseId, result]) => {
|
|
3304
|
+
total++;
|
|
3305
|
+
if (result.status === "pending" || result.status === "running") {
|
|
3306
|
+
pending++;
|
|
3307
|
+
return;
|
|
3308
|
+
}
|
|
3309
|
+
if (result.status === "failed" || result.status === "cancelled") {
|
|
3310
|
+
failed++;
|
|
3311
|
+
return;
|
|
3312
|
+
}
|
|
3313
|
+
if (result.status === "completed" && result.reportId) {
|
|
3314
|
+
const report = reports[result.reportId];
|
|
3315
|
+
if (!report) {
|
|
3316
|
+
pending++;
|
|
3317
|
+
return;
|
|
3318
|
+
}
|
|
3319
|
+
if (report.metricsStatus === "pending" || report.metricsStatus === "calculating") {
|
|
3320
|
+
pending++;
|
|
3321
|
+
return;
|
|
3322
|
+
}
|
|
3323
|
+
if (report.passFailStatus === "passed") {
|
|
3324
|
+
passed++;
|
|
3325
|
+
} else {
|
|
3326
|
+
failed++;
|
|
3327
|
+
}
|
|
3328
|
+
} else {
|
|
3329
|
+
pending++;
|
|
3330
|
+
}
|
|
3331
|
+
});
|
|
3332
|
+
const passRate = total > 0 ? Math.round(passed / total * 100) : 0;
|
|
3333
|
+
return {
|
|
3334
|
+
passed,
|
|
3335
|
+
failed,
|
|
3336
|
+
pending,
|
|
3337
|
+
total,
|
|
3338
|
+
passRate
|
|
3339
|
+
};
|
|
3340
|
+
}
|
|
3341
|
+
function getReportIdsFromRun(run) {
|
|
3342
|
+
const reportIds = /* @__PURE__ */ new Set();
|
|
3343
|
+
Object.values(run.results || {}).forEach((result) => {
|
|
3344
|
+
if (result.reportId) {
|
|
3345
|
+
reportIds.add(result.reportId);
|
|
3346
|
+
}
|
|
3347
|
+
});
|
|
3348
|
+
return Array.from(reportIds);
|
|
3349
|
+
}
|
|
3350
|
+
|
|
3351
|
+
// cli/commands/benchmark.ts
|
|
3352
|
+
function findAgent2(identifier, config) {
|
|
3353
|
+
return config.agents.find(
|
|
3354
|
+
(a) => a.key === identifier || a.name.toLowerCase() === identifier.toLowerCase()
|
|
3355
|
+
);
|
|
3356
|
+
}
|
|
3357
|
+
function getDefaultModel2(agent) {
|
|
3358
|
+
return agent.models[0] || "claude-sonnet";
|
|
3359
|
+
}
|
|
3360
|
+
function isFilePath(value) {
|
|
3361
|
+
return value.toLowerCase().endsWith(".json");
|
|
3362
|
+
}
|
|
3363
|
+
function loadAndValidateTestCasesFile(filePath) {
|
|
3364
|
+
let raw;
|
|
3365
|
+
try {
|
|
3366
|
+
raw = readFileSync2(filePath, "utf-8");
|
|
3367
|
+
} catch (err) {
|
|
3368
|
+
throw new Error(`Cannot read file: ${filePath} (${err instanceof Error ? err.message : err})`);
|
|
3369
|
+
}
|
|
3370
|
+
let parsed;
|
|
3371
|
+
try {
|
|
3372
|
+
parsed = JSON.parse(raw);
|
|
3373
|
+
} catch {
|
|
3374
|
+
throw new Error(`Invalid JSON in file: ${filePath}`);
|
|
3375
|
+
}
|
|
3376
|
+
const result = validateTestCasesArrayJson(parsed);
|
|
3377
|
+
if (!result.valid || !result.data) {
|
|
3378
|
+
const msgs = result.errors.map((e) => e.path ? `${e.path}: ${e.message}` : e.message).join("\n ");
|
|
3379
|
+
throw new Error(`Validation failed for ${filePath}:
|
|
3380
|
+
${msgs}`);
|
|
3381
|
+
}
|
|
3382
|
+
return result.data;
|
|
3383
|
+
}
|
|
3384
|
+
async function fetchReportsForRun(api, run) {
|
|
3385
|
+
const reportIds = getReportIdsFromRun(run);
|
|
3386
|
+
const reportsMap = {};
|
|
3387
|
+
await Promise.all(
|
|
3388
|
+
reportIds.map(async (reportId) => {
|
|
3389
|
+
reportsMap[reportId] = await api.getReportById(reportId);
|
|
3390
|
+
})
|
|
3391
|
+
);
|
|
3392
|
+
return reportsMap;
|
|
3393
|
+
}
|
|
3394
|
+
async function runBenchmarkForAgent(api, agent, modelId, benchmark, verbose) {
|
|
3395
|
+
const results = {
|
|
3396
|
+
agent,
|
|
3397
|
+
passed: 0,
|
|
3398
|
+
failed: 0
|
|
3399
|
+
};
|
|
3400
|
+
const totalTestCases = benchmark.testCaseIds.length;
|
|
3401
|
+
const spinner = ora2(`Running ${agent.name} (0/${totalTestCases})`).start();
|
|
3402
|
+
let startedRunId;
|
|
3403
|
+
try {
|
|
3404
|
+
const completedRun = await api.executeBenchmark(
|
|
3405
|
+
benchmark.id,
|
|
3406
|
+
{
|
|
3407
|
+
name: `CLI Run - ${agent.name}`,
|
|
3408
|
+
agentKey: agent.key,
|
|
3409
|
+
modelId
|
|
3410
|
+
},
|
|
3411
|
+
(event) => {
|
|
3412
|
+
if (event.type === "started") {
|
|
3413
|
+
startedRunId = event.runId;
|
|
3414
|
+
} else if (event.type === "progress") {
|
|
3415
|
+
const current = event.currentTestCaseIndex + 1;
|
|
3416
|
+
const testCaseName = event.currentTestCase?.name || `Test ${current}`;
|
|
3417
|
+
spinner.text = `${agent.name}: ${testCaseName} (${current}/${totalTestCases})`;
|
|
3418
|
+
if (verbose && event.result) {
|
|
3419
|
+
const status = event.result.status === "completed" ? chalk3.green("\u2713") : chalk3.red("\u2717");
|
|
3420
|
+
spinner.text = `${agent.name}: ${testCaseName} ${status} (${current}/${totalTestCases})`;
|
|
3421
|
+
}
|
|
3422
|
+
}
|
|
3423
|
+
}
|
|
3424
|
+
);
|
|
3425
|
+
results.run = completedRun;
|
|
3426
|
+
const reportsMap = await fetchReportsForRun(api, completedRun);
|
|
3427
|
+
const stats = calculateRunStats(completedRun, reportsMap);
|
|
3428
|
+
results.passed = stats.passed;
|
|
3429
|
+
results.failed = stats.failed;
|
|
3430
|
+
results.reports = Object.values(reportsMap).filter((r) => r !== null);
|
|
3431
|
+
const passRate = stats.passRate;
|
|
3432
|
+
if (passRate >= 80) {
|
|
3433
|
+
spinner.succeed(
|
|
3434
|
+
`${agent.name}: ${chalk3.green(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
|
|
3435
|
+
);
|
|
3436
|
+
} else if (passRate >= 50) {
|
|
3437
|
+
spinner.warn(
|
|
3438
|
+
`${agent.name}: ${chalk3.yellow(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
|
|
3439
|
+
);
|
|
3440
|
+
} else {
|
|
3441
|
+
spinner.fail(
|
|
3442
|
+
`${agent.name}: ${chalk3.red(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
|
|
3443
|
+
);
|
|
3444
|
+
}
|
|
3445
|
+
} catch (error) {
|
|
3446
|
+
const errorMessage = error instanceof Error ? error.message : String(error);
|
|
3447
|
+
if (startedRunId) {
|
|
3448
|
+
results.runId = startedRunId;
|
|
3449
|
+
try {
|
|
3450
|
+
const run = await api.getRun(benchmark.id, startedRunId);
|
|
3451
|
+
if (run) {
|
|
3452
|
+
results.run = run;
|
|
3453
|
+
const reportsMap = await fetchReportsForRun(api, run);
|
|
3454
|
+
const stats = calculateRunStats(run, reportsMap);
|
|
3455
|
+
results.passed = stats.passed;
|
|
3456
|
+
results.failed = stats.failed;
|
|
3457
|
+
results.reports = Object.values(reportsMap).filter((r) => r !== null);
|
|
3458
|
+
if (run.status === "completed" || run.status === "cancelled") {
|
|
3459
|
+
const passRate = stats.passRate;
|
|
3460
|
+
if (passRate >= 80) {
|
|
3461
|
+
spinner.succeed(
|
|
3462
|
+
`${agent.name}: ${chalk3.green(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
|
|
3463
|
+
);
|
|
3464
|
+
} else if (passRate >= 50) {
|
|
3465
|
+
spinner.warn(
|
|
3466
|
+
`${agent.name}: ${chalk3.yellow(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
|
|
3467
|
+
);
|
|
3468
|
+
} else {
|
|
3469
|
+
spinner.fail(
|
|
3470
|
+
`${agent.name}: ${chalk3.red(`${stats.passed}/${stats.total} passed`)} (${passRate}% pass rate)`
|
|
3471
|
+
);
|
|
3472
|
+
}
|
|
3473
|
+
return results;
|
|
3474
|
+
}
|
|
3475
|
+
}
|
|
3476
|
+
} catch {
|
|
3477
|
+
}
|
|
3478
|
+
}
|
|
3479
|
+
const isStreamError = errorMessage.includes("terminated") || errorMessage.includes("network") || errorMessage.includes("stream") || errorMessage.includes("aborted");
|
|
3480
|
+
if (isStreamError && startedRunId) {
|
|
3481
|
+
spinner.warn(`${agent.name}: ${chalk3.yellow("Stream disconnected")} - server may still be processing`);
|
|
3482
|
+
console.log(chalk3.gray(` Check status: Use the UI to monitor progress`));
|
|
3483
|
+
} else {
|
|
3484
|
+
spinner.fail(`${agent.name}: ${chalk3.red("Failed")} - ${errorMessage}`);
|
|
3485
|
+
}
|
|
3486
|
+
}
|
|
3487
|
+
return results;
|
|
3488
|
+
}
|
|
3489
|
+
function displaySummaryTable(allResults, totalTestCases) {
|
|
3490
|
+
const table = new Table3({
|
|
3491
|
+
head: [
|
|
3492
|
+
chalk3.cyan("Agent"),
|
|
3493
|
+
chalk3.cyan("Passed"),
|
|
3494
|
+
chalk3.cyan("Failed"),
|
|
3495
|
+
chalk3.cyan("Pass Rate"),
|
|
3496
|
+
chalk3.cyan("Run ID")
|
|
3497
|
+
],
|
|
3498
|
+
colWidths: [25, 10, 10, 12, 35]
|
|
3499
|
+
});
|
|
3500
|
+
for (const results of allResults) {
|
|
3501
|
+
const passRate = totalTestCases > 0 ? results.passed / totalTestCases * 100 : 0;
|
|
3502
|
+
const passRateColor = passRate >= 80 ? chalk3.green : passRate >= 50 ? chalk3.yellow : chalk3.red;
|
|
3503
|
+
table.push([
|
|
3504
|
+
results.agent.name,
|
|
3505
|
+
chalk3.green(results.passed.toString()),
|
|
3506
|
+
chalk3.red(results.failed.toString()),
|
|
3507
|
+
passRateColor(`${passRate.toFixed(0)}%`),
|
|
3508
|
+
results.run?.id || results.runId || chalk3.gray("N/A")
|
|
3509
|
+
]);
|
|
3510
|
+
}
|
|
3511
|
+
console.log("\n");
|
|
3512
|
+
console.log(chalk3.bold("Benchmark Summary"));
|
|
3513
|
+
console.log(table.toString());
|
|
3514
|
+
}
|
|
3515
|
+
async function exportResults(benchmark, allResults, exportPath, format, serverBaseUrl) {
|
|
3516
|
+
if (format !== "json") {
|
|
3517
|
+
const runIds = allResults.map((r) => r.run?.id || r.runId).filter((id) => !!id);
|
|
3518
|
+
const params = new URLSearchParams({ format });
|
|
3519
|
+
if (runIds.length > 0) {
|
|
3520
|
+
params.set("runIds", runIds.join(","));
|
|
3521
|
+
}
|
|
3522
|
+
const url = `${serverBaseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmark.id)}/report?${params.toString()}`;
|
|
3523
|
+
const response = await fetch(url);
|
|
3524
|
+
if (!response.ok) {
|
|
3525
|
+
const errorBody = await response.json().catch(() => ({ error: "Unknown error" }));
|
|
3526
|
+
console.error(chalk3.red(`
|
|
3527
|
+
Export failed: ${errorBody.error}`));
|
|
3528
|
+
return;
|
|
3529
|
+
}
|
|
3530
|
+
const contentType = response.headers.get("content-type") || "";
|
|
3531
|
+
if (contentType.includes("application/pdf")) {
|
|
3532
|
+
const buffer = Buffer.from(await response.arrayBuffer());
|
|
3533
|
+
writeFileSync(exportPath, buffer);
|
|
3534
|
+
} else {
|
|
3535
|
+
const text = await response.text();
|
|
3536
|
+
writeFileSync(exportPath, text);
|
|
3537
|
+
}
|
|
3538
|
+
} else {
|
|
3539
|
+
const exportData = {
|
|
3540
|
+
benchmark: {
|
|
3541
|
+
id: benchmark.id,
|
|
3542
|
+
name: benchmark.name,
|
|
3543
|
+
testCaseCount: benchmark.testCaseIds.length
|
|
3544
|
+
},
|
|
3545
|
+
runs: allResults.map((r) => ({
|
|
3546
|
+
agent: { key: r.agent.key, name: r.agent.name },
|
|
3547
|
+
runId: r.run?.id || r.runId,
|
|
3548
|
+
status: r.run?.status,
|
|
3549
|
+
passed: r.passed,
|
|
3550
|
+
failed: r.failed,
|
|
3551
|
+
passRate: benchmark.testCaseIds.length > 0 ? r.passed / benchmark.testCaseIds.length * 100 : 0,
|
|
3552
|
+
results: r.run?.results,
|
|
3553
|
+
reports: r.reports
|
|
3554
|
+
})),
|
|
3555
|
+
exportedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
3556
|
+
};
|
|
3557
|
+
writeFileSync(exportPath, JSON.stringify(exportData, null, 2));
|
|
3558
|
+
}
|
|
3559
|
+
console.log(chalk3.green(`
|
|
3560
|
+
Results exported to: ${exportPath}`));
|
|
3561
|
+
}
|
|
3562
|
+
function createBenchmarkCommand() {
|
|
3563
|
+
const command = new Command3("benchmark").description("Run a benchmark against one or more agents").option("-n, --name <name>", "Benchmark name or ID (optional in quick mode)").option("-f, --file <path>", "JSON file of test cases to import and benchmark").option(
|
|
3564
|
+
"-a, --agent <key>",
|
|
3565
|
+
"Agent key (can be specified multiple times)",
|
|
3566
|
+
(val, arr) => [...arr, val],
|
|
3567
|
+
[]
|
|
3568
|
+
).option("-m, --model <id>", "Model ID (uses agent default if not specified)").option("-o, --output <format>", "Output format: table, json", "table").option("--export <path>", "Export results to file").option("--format <type>", "Report format for --export: json (default), html, pdf", "json").option("-v, --verbose", "Show detailed output").option("--stop-server", "Stop the server after benchmark completes (default: keep running)").action(async (options) => {
|
|
3569
|
+
console.log(chalk3.bold("\nAgent Health - Benchmark Runner\n"));
|
|
3570
|
+
const config = await loadConfig();
|
|
3571
|
+
const serverConfig = { ...DEFAULT_SERVER_CONFIG, ...config.server };
|
|
3572
|
+
const isCI = !!process.env.CI;
|
|
3573
|
+
const serverWasRunning = await isServerRunning(serverConfig.port);
|
|
3574
|
+
const filePath = options.file || (options.name && isFilePath(options.name) ? options.name : void 0);
|
|
3575
|
+
const fileMode = !!filePath;
|
|
3576
|
+
const quickMode = !options.name && !fileMode && !serverWasRunning;
|
|
3577
|
+
if (!options.name && !fileMode && serverWasRunning) {
|
|
3578
|
+
console.error(chalk3.red(" Error: Benchmark name required when server is already running."));
|
|
3579
|
+
console.log("");
|
|
3580
|
+
console.log(chalk3.cyan(" Options:"));
|
|
3581
|
+
console.log(chalk3.gray(' 1. Specify a benchmark: benchmark -n "Name" -a claude-code'));
|
|
3582
|
+
console.log(chalk3.gray(" 2. Import from file: benchmark -f ./test-cases.json -a mock"));
|
|
3583
|
+
console.log(chalk3.gray(" 3. Stop the server and run in quick mode"));
|
|
3584
|
+
console.log(chalk3.gray(" 4. List available: npx agent-health list benchmarks"));
|
|
3585
|
+
console.log("");
|
|
3586
|
+
process.exit(1);
|
|
3587
|
+
}
|
|
3588
|
+
if (fileMode) {
|
|
3589
|
+
console.log(chalk3.cyan(` Running in file mode (importing test cases from ${filePath})`));
|
|
3590
|
+
} else if (quickMode) {
|
|
3591
|
+
console.log(chalk3.cyan(" Running in quick mode (auto-creating benchmark from test cases)"));
|
|
3592
|
+
}
|
|
3593
|
+
const connectSpinner = ora2("Connecting to server...").start();
|
|
3594
|
+
let serverResult;
|
|
3595
|
+
let cleanup;
|
|
3596
|
+
const shouldStopServer = isCI || quickMode || fileMode || options.stopServer;
|
|
3597
|
+
try {
|
|
3598
|
+
serverResult = await ensureServer(serverConfig);
|
|
3599
|
+
cleanup = createServerCleanup(serverResult, shouldStopServer);
|
|
3600
|
+
if (serverResult.wasStarted) {
|
|
3601
|
+
connectSpinner.succeed(`Started server on port ${serverConfig.port}`);
|
|
3602
|
+
} else {
|
|
3603
|
+
connectSpinner.succeed(`Connected to existing server on port ${serverConfig.port}`);
|
|
3604
|
+
}
|
|
3605
|
+
} catch (error) {
|
|
3606
|
+
connectSpinner.fail(
|
|
3607
|
+
`Failed to connect to server: ${error instanceof Error ? error.message : error}`
|
|
3608
|
+
);
|
|
3609
|
+
process.exit(1);
|
|
3610
|
+
}
|
|
3611
|
+
const api = new ApiClient(serverResult.baseUrl);
|
|
3612
|
+
try {
|
|
3613
|
+
let benchmark = null;
|
|
3614
|
+
if (fileMode) {
|
|
3615
|
+
const importSpinner = ora2(`Loading test cases from ${filePath}...`).start();
|
|
3616
|
+
try {
|
|
3617
|
+
const validatedTestCases = loadAndValidateTestCasesFile(filePath);
|
|
3618
|
+
importSpinner.succeed(`Validated ${validatedTestCases.length} test cases from file`);
|
|
3619
|
+
const uploadSpinner = ora2("Importing test cases to server...").start();
|
|
3620
|
+
const bulkResult = await api.bulkCreateTestCases(validatedTestCases);
|
|
3621
|
+
uploadSpinner.succeed(`Imported ${bulkResult.created} test cases`);
|
|
3622
|
+
const benchmarkName = options.file && options.name ? options.name : `file-${Date.now()}`;
|
|
3623
|
+
const createSpinner = ora2("Creating benchmark...").start();
|
|
3624
|
+
benchmark = await api.createBenchmark({
|
|
3625
|
+
name: benchmarkName,
|
|
3626
|
+
description: `Imported from ${filePath}`,
|
|
3627
|
+
testCaseIds: bulkResult.testCases.map((tc) => tc.id)
|
|
3628
|
+
});
|
|
3629
|
+
createSpinner.succeed(`Created benchmark: ${benchmark.name}`);
|
|
3630
|
+
} catch (error) {
|
|
3631
|
+
importSpinner.fail(`File import failed: ${error instanceof Error ? error.message : error}`);
|
|
3632
|
+
process.exit(1);
|
|
3633
|
+
}
|
|
3634
|
+
} else if (quickMode) {
|
|
3635
|
+
const testCasesSpinner = ora2("Fetching test cases...").start();
|
|
3636
|
+
try {
|
|
3637
|
+
const testCases = await api.listTestCases();
|
|
3638
|
+
if (testCases.length === 0) {
|
|
3639
|
+
testCasesSpinner.fail("No test cases found");
|
|
3640
|
+
console.log(chalk3.gray(" Add test cases via the UI or provide a file with -f option."));
|
|
3641
|
+
process.exit(1);
|
|
3642
|
+
}
|
|
3643
|
+
testCasesSpinner.succeed(`Found ${testCases.length} test cases`);
|
|
3644
|
+
const createSpinner = ora2("Creating quick benchmark...").start();
|
|
3645
|
+
benchmark = await api.createBenchmark({
|
|
3646
|
+
name: `quick-${Date.now()}`,
|
|
3647
|
+
description: "Auto-generated benchmark for quick mode",
|
|
3648
|
+
testCaseIds: testCases.map((tc) => tc.id)
|
|
3649
|
+
});
|
|
3650
|
+
createSpinner.succeed(`Created benchmark: ${benchmark.name}`);
|
|
3651
|
+
} catch (error) {
|
|
3652
|
+
testCasesSpinner.fail(`Failed to create benchmark: ${error instanceof Error ? error.message : error}`);
|
|
3653
|
+
process.exit(1);
|
|
3654
|
+
}
|
|
3655
|
+
} else {
|
|
3656
|
+
benchmark = await api.findBenchmark(options.name);
|
|
3657
|
+
if (!benchmark) {
|
|
3658
|
+
console.error(chalk3.red(` Error: Benchmark not found: "${options.name}"`));
|
|
3659
|
+
console.log("");
|
|
3660
|
+
console.log(chalk3.cyan(" The -n/--name option accepts:"));
|
|
3661
|
+
console.log(chalk3.gray(" \u2022 Benchmark ID (e.g., demo-baseline)"));
|
|
3662
|
+
console.log(chalk3.gray(' \u2022 Benchmark name (case-sensitive, e.g., "Baseline")'));
|
|
3663
|
+
console.log("");
|
|
3664
|
+
console.log(chalk3.cyan(" Or import from file:"));
|
|
3665
|
+
console.log(chalk3.gray(" benchmark -f ./test-cases.json -a mock"));
|
|
3666
|
+
console.log("");
|
|
3667
|
+
console.log(chalk3.cyan(" Available benchmarks:"));
|
|
3668
|
+
console.log(chalk3.gray(" npx agent-health list benchmarks"));
|
|
3669
|
+
console.log("");
|
|
3670
|
+
process.exit(1);
|
|
3671
|
+
}
|
|
3672
|
+
if (benchmark.id.startsWith("demo-")) {
|
|
3673
|
+
console.error(chalk3.red(` Error: Cannot execute sample benchmarks.`));
|
|
3674
|
+
console.log(chalk3.gray(" Sample data is read-only with pre-completed runs."));
|
|
3675
|
+
console.log(chalk3.gray(" Create a real benchmark in the UI to run evaluations."));
|
|
3676
|
+
console.log("");
|
|
3677
|
+
process.exit(1);
|
|
3678
|
+
}
|
|
3679
|
+
}
|
|
3680
|
+
console.log(chalk3.gray(` Benchmark: ${benchmark.name} (${benchmark.id})`));
|
|
3681
|
+
console.log(chalk3.gray(` Test Cases: ${benchmark.testCaseIds.length}`));
|
|
3682
|
+
console.log(chalk3.gray(` Server: ${serverResult.baseUrl}`));
|
|
3683
|
+
let agents = [];
|
|
3684
|
+
if (options.agent.length === 0) {
|
|
3685
|
+
const enabledAgent = config.agents.find((a) => a.enabled !== false);
|
|
3686
|
+
if (!enabledAgent) {
|
|
3687
|
+
console.error(chalk3.red(" Error: No enabled agents found in config."));
|
|
3688
|
+
process.exit(1);
|
|
3689
|
+
}
|
|
3690
|
+
agents = [enabledAgent];
|
|
3691
|
+
console.log(chalk3.gray(` Agent: ${agents[0].name} (default)`));
|
|
3692
|
+
} else {
|
|
3693
|
+
for (const agentId of options.agent) {
|
|
3694
|
+
const agent = findAgent2(agentId, config);
|
|
3695
|
+
if (!agent) {
|
|
3696
|
+
console.error(chalk3.red(` Error: Agent not found: ${agentId}`));
|
|
3697
|
+
console.log(chalk3.gray(" Available agents:"));
|
|
3698
|
+
for (const a of config.agents) {
|
|
3699
|
+
console.log(chalk3.gray(` - ${a.name} (${a.key})`));
|
|
3700
|
+
}
|
|
3701
|
+
console.log("");
|
|
3702
|
+
process.exit(1);
|
|
3703
|
+
}
|
|
3704
|
+
agents.push(agent);
|
|
3705
|
+
}
|
|
3706
|
+
console.log(chalk3.gray(` Agents: ${agents.map((a) => a.name).join(", ")}`));
|
|
3707
|
+
}
|
|
3708
|
+
console.log("");
|
|
3709
|
+
const allResults = [];
|
|
3710
|
+
for (const agent of agents) {
|
|
3711
|
+
const modelId = options.model || getDefaultModel2(agent);
|
|
3712
|
+
const results = await runBenchmarkForAgent(
|
|
3713
|
+
api,
|
|
3714
|
+
agent,
|
|
3715
|
+
modelId,
|
|
3716
|
+
benchmark,
|
|
3717
|
+
options.verbose || false
|
|
3718
|
+
);
|
|
3719
|
+
allResults.push(results);
|
|
3720
|
+
}
|
|
3721
|
+
if (options.output === "json") {
|
|
3722
|
+
const jsonOutput = allResults.map((r) => ({
|
|
3723
|
+
agent: { key: r.agent.key, name: r.agent.name },
|
|
3724
|
+
runId: r.run?.id || r.runId,
|
|
3725
|
+
passed: r.passed,
|
|
3726
|
+
failed: r.failed,
|
|
3727
|
+
passRate: benchmark.testCaseIds.length > 0 ? r.passed / benchmark.testCaseIds.length * 100 : 0,
|
|
3728
|
+
results: r.run?.results
|
|
3729
|
+
}));
|
|
3730
|
+
console.log(JSON.stringify(jsonOutput, null, 2));
|
|
3731
|
+
} else {
|
|
3732
|
+
displaySummaryTable(allResults, benchmark.testCaseIds.length);
|
|
3733
|
+
}
|
|
3734
|
+
if (options.export) {
|
|
3735
|
+
await exportResults(benchmark, allResults, options.export, options.format, serverResult.baseUrl);
|
|
3736
|
+
}
|
|
3737
|
+
console.log("");
|
|
3738
|
+
console.log(chalk3.cyan("View results:"));
|
|
3739
|
+
for (const result of allResults) {
|
|
3740
|
+
const runId = result.run?.id || result.runId;
|
|
3741
|
+
if (runId) {
|
|
3742
|
+
console.log(chalk3.gray(` ${result.agent.name}: ${serverResult.baseUrl}/benchmarks/${benchmark.id}/runs/${runId}`));
|
|
3743
|
+
}
|
|
3744
|
+
}
|
|
3745
|
+
if (process.env.OPENSEARCH_DASHBOARDS_URL) {
|
|
3746
|
+
console.log(chalk3.gray(` OpenSearch Dashboards: ${process.env.OPENSEARCH_DASHBOARDS_URL}`));
|
|
3747
|
+
}
|
|
3748
|
+
if (serverResult.wasStarted && !shouldStopServer) {
|
|
3749
|
+
console.log("");
|
|
3750
|
+
console.log(chalk3.gray(`Server still running on port ${serverConfig.port}`));
|
|
3751
|
+
console.log(chalk3.gray(` Use --stop-server flag to stop after benchmark`));
|
|
3752
|
+
console.log(chalk3.gray(` Or manually: kill $(lsof -t -i:${serverConfig.port})`));
|
|
3753
|
+
}
|
|
3754
|
+
} finally {
|
|
3755
|
+
cleanup();
|
|
3756
|
+
}
|
|
3757
|
+
});
|
|
3758
|
+
return command;
|
|
3759
|
+
}
|
|
3760
|
+
|
|
3761
|
+
// cli/commands/export.ts
|
|
3762
|
+
import { Command as Command4 } from "commander";
|
|
3763
|
+
import chalk4 from "chalk";
|
|
3764
|
+
import { writeFileSync as writeFileSync2 } from "fs";
|
|
3765
|
+
|
|
3766
|
+
// lib/benchmarkExport.ts
|
|
3767
|
+
function generateExportFilename(benchmarkName) {
|
|
3768
|
+
const sanitized = (benchmarkName || "benchmark-export").trim().toLowerCase().replace(/[^a-z0-9_\-\s]/g, "").replace(/\s+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "");
|
|
3769
|
+
return `${sanitized || "benchmark-export"}.json`;
|
|
3770
|
+
}
|
|
3771
|
+
|
|
3772
|
+
// cli/commands/export.ts
|
|
3773
|
+
function createExportCommand() {
|
|
3774
|
+
const command = new Command4("export").description("Export benchmark test cases as JSON").requiredOption("-b, --benchmark <id-or-name>", "Benchmark ID or name").option("-o, --output <file>", "Output file path (default: <benchmark-name>.json)").option("--stdout", "Write to stdout instead of file").action(async (options) => {
|
|
3775
|
+
const config = await loadConfig();
|
|
3776
|
+
const serverResult = await ensureServer(config.server);
|
|
3777
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
3778
|
+
try {
|
|
3779
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
3780
|
+
const benchmark = await client.findBenchmark(options.benchmark);
|
|
3781
|
+
if (!benchmark) {
|
|
3782
|
+
console.error(chalk4.red(`
|
|
3783
|
+
Error: Benchmark not found: ${options.benchmark}
|
|
3784
|
+
`));
|
|
3785
|
+
process.exit(1);
|
|
3786
|
+
}
|
|
3787
|
+
const exportData = await client.exportBenchmark(benchmark.id);
|
|
3788
|
+
if (options.stdout) {
|
|
3789
|
+
process.stdout.write(JSON.stringify(exportData, null, 2) + "\n");
|
|
3790
|
+
return;
|
|
3791
|
+
}
|
|
3792
|
+
const outputFile = options.output || generateExportFilename(benchmark.name);
|
|
3793
|
+
writeFileSync2(outputFile, JSON.stringify(exportData, null, 2) + "\n", "utf-8");
|
|
3794
|
+
console.log(chalk4.green(`
|
|
3795
|
+
Exported ${exportData.length} test case(s) to ${chalk4.bold(outputFile)}`));
|
|
3796
|
+
console.log(chalk4.gray(` Benchmark: ${benchmark.name} (${benchmark.id})
|
|
3797
|
+
`));
|
|
3798
|
+
} catch (error) {
|
|
3799
|
+
console.error(chalk4.red(`
|
|
3800
|
+
Error: ${error.message}`));
|
|
3801
|
+
console.log(chalk4.gray(" Is the server running? Start with: npm run dev:server\n"));
|
|
3802
|
+
process.exit(1);
|
|
3803
|
+
} finally {
|
|
3804
|
+
cleanup();
|
|
3805
|
+
}
|
|
3806
|
+
});
|
|
3807
|
+
return command;
|
|
3808
|
+
}
|
|
3809
|
+
|
|
3810
|
+
// cli/commands/report.ts
|
|
3811
|
+
import { Command as Command5 } from "commander";
|
|
3812
|
+
import chalk5 from "chalk";
|
|
3813
|
+
import ora3 from "ora";
|
|
3814
|
+
import { writeFileSync as writeFileSync3 } from "fs";
|
|
3815
|
+
function createReportCommand() {
|
|
3816
|
+
const command = new Command5("report").description("Generate a report for a benchmark").requiredOption("-b, --benchmark <id>", "Benchmark name or ID").option("-r, --runs <ids>", "Comma-separated run IDs (default: all runs)").option("-f, --format <type>", "Report format: json, html, pdf", "html").option("-o, --output <file>", "Output file path (auto-generates filename if omitted)").option("--stdout", "Write to stdout (JSON format only)").action(async (options) => {
|
|
3817
|
+
const config = await loadConfig();
|
|
3818
|
+
const serverConfig = { ...DEFAULT_SERVER_CONFIG, ...config.server };
|
|
3819
|
+
const connectSpinner = ora3("Connecting to server...").start();
|
|
3820
|
+
let serverResult;
|
|
3821
|
+
let cleanup;
|
|
3822
|
+
try {
|
|
3823
|
+
serverResult = await ensureServer(serverConfig);
|
|
3824
|
+
cleanup = createServerCleanup(serverResult, false);
|
|
3825
|
+
if (serverResult.wasStarted) {
|
|
3826
|
+
connectSpinner.succeed(`Started server on port ${serverConfig.port}`);
|
|
3827
|
+
} else {
|
|
3828
|
+
connectSpinner.succeed(`Connected to existing server on port ${serverConfig.port}`);
|
|
3829
|
+
}
|
|
3830
|
+
} catch (error) {
|
|
3831
|
+
connectSpinner.fail(
|
|
3832
|
+
`Failed to connect to server: ${error instanceof Error ? error.message : error}`
|
|
3833
|
+
);
|
|
3834
|
+
process.exit(1);
|
|
3835
|
+
}
|
|
3836
|
+
const api = new ApiClient(serverResult.baseUrl);
|
|
3837
|
+
try {
|
|
3838
|
+
const spinner = ora3("Finding benchmark...").start();
|
|
3839
|
+
const benchmark = await api.findBenchmark(options.benchmark);
|
|
3840
|
+
if (!benchmark) {
|
|
3841
|
+
spinner.fail(`Benchmark not found: "${options.benchmark}"`);
|
|
3842
|
+
console.log("");
|
|
3843
|
+
console.log(chalk5.cyan(" Available benchmarks:"));
|
|
3844
|
+
console.log(chalk5.gray(" npx agent-health list benchmarks"));
|
|
3845
|
+
console.log("");
|
|
3846
|
+
process.exit(1);
|
|
3847
|
+
}
|
|
3848
|
+
spinner.succeed(`Found benchmark: ${benchmark.name} (${benchmark.id})`);
|
|
3849
|
+
const params = new URLSearchParams({ format: options.format });
|
|
3850
|
+
if (options.runs) {
|
|
3851
|
+
params.set("runIds", options.runs);
|
|
3852
|
+
}
|
|
3853
|
+
const reportSpinner = ora3(`Generating ${options.format.toUpperCase()} report...`).start();
|
|
3854
|
+
const url = `${serverResult.baseUrl}/api/storage/benchmarks/${encodeURIComponent(benchmark.id)}/report?${params.toString()}`;
|
|
3855
|
+
const response = await fetch(url);
|
|
3856
|
+
if (!response.ok) {
|
|
3857
|
+
const errorBody = await response.json().catch(() => ({ error: "Unknown error" }));
|
|
3858
|
+
reportSpinner.fail(`Report generation failed: ${errorBody.error}`);
|
|
3859
|
+
process.exit(1);
|
|
3860
|
+
}
|
|
3861
|
+
const contentDisposition = response.headers.get("content-disposition") || "";
|
|
3862
|
+
const filenameMatch = contentDisposition.match(/filename="([^"]+)"/);
|
|
3863
|
+
const defaultFilename = filenameMatch?.[1] || `report.${options.format}`;
|
|
3864
|
+
if (options.stdout) {
|
|
3865
|
+
const text = await response.text();
|
|
3866
|
+
reportSpinner.stop();
|
|
3867
|
+
process.stdout.write(text);
|
|
3868
|
+
} else {
|
|
3869
|
+
const outputPath = options.output || defaultFilename;
|
|
3870
|
+
const contentType = response.headers.get("content-type") || "";
|
|
3871
|
+
if (contentType.includes("application/pdf")) {
|
|
3872
|
+
const buffer = Buffer.from(await response.arrayBuffer());
|
|
3873
|
+
writeFileSync3(outputPath, buffer);
|
|
3874
|
+
} else {
|
|
3875
|
+
const text = await response.text();
|
|
3876
|
+
writeFileSync3(outputPath, text);
|
|
3877
|
+
}
|
|
3878
|
+
reportSpinner.succeed(`Report saved to: ${outputPath}`);
|
|
3879
|
+
}
|
|
3880
|
+
} finally {
|
|
3881
|
+
cleanup();
|
|
3882
|
+
}
|
|
3883
|
+
});
|
|
3884
|
+
return command;
|
|
3885
|
+
}
|
|
3886
|
+
|
|
3887
|
+
// cli/commands/doctor.ts
|
|
3888
|
+
import { Command as Command6 } from "commander";
|
|
3889
|
+
import chalk6 from "chalk";
|
|
3890
|
+
import { existsSync as existsSync3 } from "fs";
|
|
3891
|
+
import { resolve as resolve2 } from "path";
|
|
3892
|
+
function checkConfigFile() {
|
|
3893
|
+
const configInfo = getConfigFileInfo();
|
|
3894
|
+
if (configInfo) {
|
|
3895
|
+
return {
|
|
3896
|
+
name: "Config File",
|
|
3897
|
+
status: "ok",
|
|
3898
|
+
message: `Found: ${configInfo.path.split("/").pop()}`,
|
|
3899
|
+
details: [`Format: ${configInfo.format}`]
|
|
3900
|
+
};
|
|
3901
|
+
}
|
|
3902
|
+
return {
|
|
3903
|
+
name: "Config File",
|
|
3904
|
+
status: "ok",
|
|
3905
|
+
message: "Using defaults + environment variables",
|
|
3906
|
+
details: [
|
|
3907
|
+
"Config file is optional (for custom agents)",
|
|
3908
|
+
"Run `agent-health init` to create one if needed"
|
|
3909
|
+
]
|
|
3910
|
+
};
|
|
3911
|
+
}
|
|
3912
|
+
function checkEnvFile() {
|
|
3913
|
+
const envPath = resolve2(process.cwd(), ".env");
|
|
3914
|
+
if (existsSync3(envPath)) {
|
|
3915
|
+
return {
|
|
3916
|
+
name: "Environment File",
|
|
3917
|
+
status: "ok",
|
|
3918
|
+
message: "Found: .env"
|
|
3919
|
+
};
|
|
3920
|
+
}
|
|
3921
|
+
return {
|
|
3922
|
+
name: "Environment File",
|
|
3923
|
+
status: "ok",
|
|
3924
|
+
message: "Not using .env file",
|
|
3925
|
+
details: ["Env vars can be set in shell, CI/CD, or via --env-file"]
|
|
3926
|
+
};
|
|
3927
|
+
}
|
|
3928
|
+
function checkAWSCredentials() {
|
|
3929
|
+
const profile = process.env.AWS_PROFILE;
|
|
3930
|
+
const accessKey = process.env.AWS_ACCESS_KEY_ID;
|
|
3931
|
+
const region = process.env.AWS_REGION || process.env.AWS_DEFAULT_REGION;
|
|
3932
|
+
const details = [];
|
|
3933
|
+
if (profile) {
|
|
3934
|
+
details.push(`AWS_PROFILE: ${profile}`);
|
|
3935
|
+
}
|
|
3936
|
+
if (accessKey) {
|
|
3937
|
+
details.push(`AWS_ACCESS_KEY_ID: ${accessKey.substring(0, 8)}...`);
|
|
3938
|
+
}
|
|
3939
|
+
if (region) {
|
|
3940
|
+
details.push(`AWS_REGION: ${region}`);
|
|
3941
|
+
}
|
|
3942
|
+
if (profile || accessKey) {
|
|
3943
|
+
return {
|
|
3944
|
+
name: "AWS Credentials",
|
|
3945
|
+
status: "ok",
|
|
3946
|
+
message: profile ? `Profile: ${profile}` : "Using access key",
|
|
3947
|
+
details: details.length > 0 ? details : void 0
|
|
3948
|
+
};
|
|
3949
|
+
}
|
|
3950
|
+
return {
|
|
3951
|
+
name: "AWS Credentials",
|
|
3952
|
+
status: "warning",
|
|
3953
|
+
message: "No AWS credentials detected",
|
|
3954
|
+
details: [
|
|
3955
|
+
"Set AWS_PROFILE or AWS_ACCESS_KEY_ID for Bedrock judge",
|
|
3956
|
+
"Claude Code connector also requires AWS credentials"
|
|
3957
|
+
]
|
|
3958
|
+
};
|
|
3959
|
+
}
|
|
3960
|
+
async function checkClaudeCodeCLI() {
|
|
3961
|
+
const { execSync: execSync2 } = await import("child_process");
|
|
3962
|
+
try {
|
|
3963
|
+
execSync2("which claude", { stdio: "pipe" });
|
|
3964
|
+
return {
|
|
3965
|
+
name: "Claude Code CLI",
|
|
3966
|
+
status: "ok",
|
|
3967
|
+
message: "claude command available"
|
|
3968
|
+
};
|
|
3969
|
+
} catch {
|
|
3970
|
+
return {
|
|
3971
|
+
name: "Claude Code CLI",
|
|
3972
|
+
status: "warning",
|
|
3973
|
+
message: "claude command not found",
|
|
3974
|
+
details: [
|
|
3975
|
+
"Install Claude Code: npm install -g @anthropic/claude-code",
|
|
3976
|
+
"Or skip if not using claude-code connector"
|
|
3977
|
+
]
|
|
3978
|
+
};
|
|
3979
|
+
}
|
|
3980
|
+
}
|
|
3981
|
+
function checkAgents(config) {
|
|
3982
|
+
const agents = config.agents;
|
|
3983
|
+
const details = [];
|
|
3984
|
+
for (const agent of agents) {
|
|
3985
|
+
const connectorType = agent.connectorType || "agui-streaming";
|
|
3986
|
+
const hasConnector = connectorRegistry.get(connectorType);
|
|
3987
|
+
const status = hasConnector ? "\u2713" : "\u2717";
|
|
3988
|
+
details.push(`${status} ${agent.name} (${connectorType})`);
|
|
3989
|
+
}
|
|
3990
|
+
return {
|
|
3991
|
+
name: "Agents",
|
|
3992
|
+
status: "ok",
|
|
3993
|
+
message: `${agents.length} agents configured`,
|
|
3994
|
+
details
|
|
3995
|
+
};
|
|
3996
|
+
}
|
|
3997
|
+
function checkConnectors() {
|
|
3998
|
+
const types = connectorRegistry.getRegisteredTypes();
|
|
3999
|
+
return {
|
|
4000
|
+
name: "Connectors",
|
|
4001
|
+
status: "ok",
|
|
4002
|
+
message: `${types.length} connectors registered`,
|
|
4003
|
+
details: types.map((t) => `\u2713 ${t}`)
|
|
4004
|
+
};
|
|
4005
|
+
}
|
|
4006
|
+
function checkOpenSearchStorage() {
|
|
4007
|
+
const endpoint = process.env.OPENSEARCH_STORAGE_ENDPOINT;
|
|
4008
|
+
const user = process.env.OPENSEARCH_STORAGE_USERNAME;
|
|
4009
|
+
if (endpoint) {
|
|
4010
|
+
return {
|
|
4011
|
+
name: "OpenSearch Storage",
|
|
4012
|
+
status: "ok",
|
|
4013
|
+
message: `Configured: ${endpoint.substring(0, 50)}...`,
|
|
4014
|
+
details: user ? [`Username: ${user}`] : void 0
|
|
4015
|
+
};
|
|
4016
|
+
}
|
|
4017
|
+
return {
|
|
4018
|
+
name: "OpenSearch Storage",
|
|
4019
|
+
status: "warning",
|
|
4020
|
+
message: "Not configured (results won't persist)",
|
|
4021
|
+
details: [
|
|
4022
|
+
"Set OPENSEARCH_STORAGE_ENDPOINT, _USERNAME, _PASSWORD to save results",
|
|
4023
|
+
"Without storage, results are shown in terminal only"
|
|
4024
|
+
]
|
|
4025
|
+
};
|
|
4026
|
+
}
|
|
4027
|
+
function checkOpenSearchObservability() {
|
|
4028
|
+
const endpoint = process.env.OPENSEARCH_LOGS_ENDPOINT;
|
|
4029
|
+
const user = process.env.OPENSEARCH_LOGS_USERNAME;
|
|
4030
|
+
if (endpoint) {
|
|
4031
|
+
return {
|
|
4032
|
+
name: "OpenSearch Observability",
|
|
4033
|
+
status: "ok",
|
|
4034
|
+
message: `Configured: ${endpoint.substring(0, 50)}...`,
|
|
4035
|
+
details: user ? [`Username: ${user}`] : void 0
|
|
4036
|
+
};
|
|
4037
|
+
}
|
|
4038
|
+
return {
|
|
4039
|
+
name: "OpenSearch Observability",
|
|
4040
|
+
status: "ok",
|
|
4041
|
+
message: "Not configured (optional)",
|
|
4042
|
+
details: [
|
|
4043
|
+
"Set OPENSEARCH_LOGS_ENDPOINT for agent traces",
|
|
4044
|
+
"Only needed for ML-Commons agent observability"
|
|
4045
|
+
]
|
|
4046
|
+
};
|
|
4047
|
+
}
|
|
4048
|
+
function displayResults(results) {
|
|
4049
|
+
console.log(chalk6.bold("\n Configuration Check\n"));
|
|
4050
|
+
for (const result of results) {
|
|
4051
|
+
const icon = {
|
|
4052
|
+
ok: chalk6.green("\u2713"),
|
|
4053
|
+
warning: chalk6.yellow("\u26A0"),
|
|
4054
|
+
error: chalk6.red("\u2717")
|
|
4055
|
+
}[result.status];
|
|
4056
|
+
const messageColor = {
|
|
4057
|
+
ok: chalk6.green,
|
|
4058
|
+
warning: chalk6.yellow,
|
|
4059
|
+
error: chalk6.red
|
|
4060
|
+
}[result.status];
|
|
4061
|
+
console.log(` ${icon} ${chalk6.bold(result.name)}: ${messageColor(result.message)}`);
|
|
4062
|
+
if (result.details) {
|
|
4063
|
+
for (const detail of result.details) {
|
|
4064
|
+
console.log(chalk6.gray(` ${detail}`));
|
|
4065
|
+
}
|
|
4066
|
+
}
|
|
4067
|
+
}
|
|
4068
|
+
console.log("");
|
|
4069
|
+
const errors = results.filter((r) => r.status === "error").length;
|
|
4070
|
+
const warnings = results.filter((r) => r.status === "warning").length;
|
|
4071
|
+
if (errors > 0) {
|
|
4072
|
+
console.log(chalk6.red(` ${errors} error(s) found. Fix these before running evaluations.
|
|
4073
|
+
`));
|
|
4074
|
+
} else if (warnings > 0) {
|
|
4075
|
+
console.log(chalk6.yellow(` ${warnings} warning(s). Some features may be limited.
|
|
4076
|
+
`));
|
|
4077
|
+
} else {
|
|
4078
|
+
console.log(chalk6.green(" All checks passed!\n"));
|
|
4079
|
+
}
|
|
4080
|
+
}
|
|
4081
|
+
function createDoctorCommand() {
|
|
4082
|
+
const command = new Command6("doctor").description("Check configuration and system requirements").option("-o, --output <format>", "Output format: text, json", "text").action(async (options) => {
|
|
4083
|
+
const results = [];
|
|
4084
|
+
const config = await loadConfig();
|
|
4085
|
+
for (const connector of config.connectors) {
|
|
4086
|
+
connectorRegistry.register(connector);
|
|
4087
|
+
}
|
|
4088
|
+
results.push(checkConfigFile());
|
|
4089
|
+
results.push(checkEnvFile());
|
|
4090
|
+
results.push(checkAWSCredentials());
|
|
4091
|
+
results.push(await checkClaudeCodeCLI());
|
|
4092
|
+
results.push(checkAgents(config));
|
|
4093
|
+
results.push(checkConnectors());
|
|
4094
|
+
results.push(checkOpenSearchStorage());
|
|
4095
|
+
results.push(checkOpenSearchObservability());
|
|
4096
|
+
if (options.output === "json") {
|
|
4097
|
+
console.log(JSON.stringify(results, null, 2));
|
|
4098
|
+
} else {
|
|
4099
|
+
displayResults(results);
|
|
4100
|
+
}
|
|
4101
|
+
});
|
|
4102
|
+
return command;
|
|
4103
|
+
}
|
|
4104
|
+
|
|
4105
|
+
// cli/commands/init.ts
|
|
4106
|
+
import { Command as Command7 } from "commander";
|
|
4107
|
+
import chalk7 from "chalk";
|
|
4108
|
+
import { writeFileSync as writeFileSync4, existsSync as existsSync4 } from "fs";
|
|
4109
|
+
import { resolve as resolve3 } from "path";
|
|
4110
|
+
var TYPESCRIPT_CONFIG = `/*
|
|
4111
|
+
* Agent Health Configuration
|
|
4112
|
+
* See docs/CONFIGURATION.md for full options
|
|
4113
|
+
*/
|
|
4114
|
+
|
|
4115
|
+
import { defineConfig, AGUIStreamingConnector, ClaudeCodeConnector } from '@opensearch-project/agent-health';
|
|
4116
|
+
|
|
4117
|
+
export default defineConfig({
|
|
4118
|
+
agents: [
|
|
4119
|
+
// ML-Commons agent (AG-UI streaming)
|
|
4120
|
+
{
|
|
4121
|
+
name: 'ml-commons',
|
|
4122
|
+
key: 'ml-commons',
|
|
4123
|
+
connector: new AGUIStreamingConnector(),
|
|
4124
|
+
endpoint: process.env.MLCOMMONS_ENDPOINT || 'https://localhost:9200/_plugins/_ml/agents/YOUR_AGENT_ID/_execute',
|
|
4125
|
+
auth: {
|
|
4126
|
+
type: 'basic',
|
|
4127
|
+
username: process.env.OPENSEARCH_USER || 'admin',
|
|
4128
|
+
password: process.env.OPENSEARCH_PASS || 'admin',
|
|
4129
|
+
},
|
|
4130
|
+
models: ['claude-sonnet'],
|
|
4131
|
+
},
|
|
4132
|
+
|
|
4133
|
+
// Claude Code CLI agent (optional)
|
|
4134
|
+
// Uncomment to enable Claude Code comparison
|
|
4135
|
+
/*
|
|
4136
|
+
{
|
|
4137
|
+
name: 'claude-code',
|
|
4138
|
+
key: 'claude-code',
|
|
4139
|
+
connector: new ClaudeCodeConnector({
|
|
4140
|
+
env: {
|
|
4141
|
+
AWS_PROFILE: process.env.AWS_PROFILE || 'Bedrock',
|
|
4142
|
+
CLAUDE_CODE_USE_BEDROCK: '1',
|
|
4143
|
+
AWS_REGION: process.env.AWS_REGION || 'us-west-2',
|
|
4144
|
+
},
|
|
4145
|
+
}),
|
|
4146
|
+
endpoint: 'claude', // Command name
|
|
4147
|
+
models: ['claude-sonnet-4'],
|
|
4148
|
+
},
|
|
4149
|
+
*/
|
|
4150
|
+
],
|
|
4151
|
+
|
|
4152
|
+
// Test cases can be inline or loaded from files
|
|
4153
|
+
testCases: './test-cases/*.yaml',
|
|
4154
|
+
|
|
4155
|
+
// Output reporters
|
|
4156
|
+
reporters: [
|
|
4157
|
+
['console'],
|
|
4158
|
+
['json', { output: 'report.json' }],
|
|
4159
|
+
],
|
|
4160
|
+
|
|
4161
|
+
// Judge configuration
|
|
4162
|
+
judge: {
|
|
4163
|
+
provider: 'bedrock',
|
|
4164
|
+
model: 'claude-sonnet',
|
|
4165
|
+
region: process.env.AWS_REGION || 'us-west-2',
|
|
4166
|
+
},
|
|
4167
|
+
});
|
|
4168
|
+
`;
|
|
4169
|
+
var ENV_TEMPLATE = `# Agent Health Environment Configuration
|
|
4170
|
+
# Copy this to .env and fill in your values
|
|
4171
|
+
|
|
4172
|
+
# ============ AWS Configuration ============
|
|
4173
|
+
# For Bedrock judge and Claude Code CLI
|
|
4174
|
+
AWS_PROFILE=Bedrock
|
|
4175
|
+
AWS_REGION=us-west-2
|
|
4176
|
+
# Or use explicit credentials:
|
|
4177
|
+
# AWS_ACCESS_KEY_ID=
|
|
4178
|
+
# AWS_SECRET_ACCESS_KEY=
|
|
4179
|
+
# AWS_SESSION_TOKEN=
|
|
4180
|
+
|
|
4181
|
+
# ============ OpenSearch/ML-Commons ============
|
|
4182
|
+
# Agent endpoint
|
|
4183
|
+
MLCOMMONS_ENDPOINT=https://localhost:9200/_plugins/_ml/agents/YOUR_AGENT_ID/_execute
|
|
4184
|
+
|
|
4185
|
+
# Storage cluster (for test cases, benchmarks persistence)
|
|
4186
|
+
OPENSEARCH_STORAGE_URL=https://localhost:9200
|
|
4187
|
+
OPENSEARCH_STORAGE_USER=admin
|
|
4188
|
+
OPENSEARCH_STORAGE_PASS=admin
|
|
4189
|
+
|
|
4190
|
+
# Optional: Headers for ML-Commons agent data source access
|
|
4191
|
+
# MLCOMMONS_HEADER_OPENSEARCH_URL=
|
|
4192
|
+
# MLCOMMONS_HEADER_AUTHORIZATION=
|
|
4193
|
+
|
|
4194
|
+
# ============ Server Configuration ============
|
|
4195
|
+
# Backend port (default: 4001)
|
|
4196
|
+
# BACKEND_PORT=4001
|
|
4197
|
+
`;
|
|
4198
|
+
var SAMPLE_TEST_CASE = `# Sample Test Case
|
|
4199
|
+
# Place in test-cases/ directory
|
|
4200
|
+
|
|
4201
|
+
id: sample-rca-001
|
|
4202
|
+
name: Sample RCA Test Case
|
|
4203
|
+
version: 1
|
|
4204
|
+
|
|
4205
|
+
labels:
|
|
4206
|
+
- category:RCA
|
|
4207
|
+
- difficulty:Medium
|
|
4208
|
+
|
|
4209
|
+
initialPrompt: |
|
|
4210
|
+
A customer reports that their web application is experiencing
|
|
4211
|
+
slow response times. The issue started approximately 2 hours ago.
|
|
4212
|
+
Please investigate and identify the root cause.
|
|
4213
|
+
|
|
4214
|
+
context:
|
|
4215
|
+
- description: Application logs
|
|
4216
|
+
value: |
|
|
4217
|
+
2024-01-15 10:00:00 ERROR: Connection timeout to database
|
|
4218
|
+
2024-01-15 10:00:05 WARN: Retry attempt 1 for DB connection
|
|
4219
|
+
2024-01-15 10:00:10 ERROR: Connection timeout to database
|
|
4220
|
+
|
|
4221
|
+
expectedOutcomes:
|
|
4222
|
+
- The agent should identify database connectivity issues
|
|
4223
|
+
- The agent should check database server health
|
|
4224
|
+
- The agent should suggest investigating network connectivity
|
|
4225
|
+
`;
|
|
4226
|
+
function createInitCommand() {
|
|
4227
|
+
const command = new Command7("init").description("Initialize configuration files").option("--force", "Overwrite existing files").option("--with-examples", "Include example test case").action(async (options) => {
|
|
4228
|
+
console.log(chalk7.bold("\n Agent Health - Initialize Configuration\n"));
|
|
4229
|
+
const cwd = process.cwd();
|
|
4230
|
+
const files = [];
|
|
4231
|
+
files.push({
|
|
4232
|
+
path: resolve3(cwd, "agent-health.config.ts"),
|
|
4233
|
+
content: TYPESCRIPT_CONFIG,
|
|
4234
|
+
name: "agent-health.config.ts"
|
|
4235
|
+
});
|
|
4236
|
+
files.push({
|
|
4237
|
+
path: resolve3(cwd, ".env.example"),
|
|
4238
|
+
content: ENV_TEMPLATE,
|
|
4239
|
+
name: ".env.example"
|
|
4240
|
+
});
|
|
4241
|
+
if (options.withExamples) {
|
|
4242
|
+
const { mkdirSync } = await import("fs");
|
|
4243
|
+
const testCasesDir = resolve3(cwd, "test-cases");
|
|
4244
|
+
if (!existsSync4(testCasesDir)) {
|
|
4245
|
+
mkdirSync(testCasesDir, { recursive: true });
|
|
4246
|
+
}
|
|
4247
|
+
files.push({
|
|
4248
|
+
path: resolve3(testCasesDir, "sample-rca.yaml"),
|
|
4249
|
+
content: SAMPLE_TEST_CASE,
|
|
4250
|
+
name: "test-cases/sample-rca.yaml"
|
|
4251
|
+
});
|
|
4252
|
+
}
|
|
4253
|
+
let created = 0;
|
|
4254
|
+
let skipped = 0;
|
|
4255
|
+
for (const file of files) {
|
|
4256
|
+
if (existsSync4(file.path) && !options.force) {
|
|
4257
|
+
console.log(chalk7.yellow(` \u26A0 Skipped: ${file.name} (already exists, use --force to overwrite)`));
|
|
4258
|
+
skipped++;
|
|
4259
|
+
} else {
|
|
4260
|
+
writeFileSync4(file.path, file.content);
|
|
4261
|
+
console.log(chalk7.green(` \u2713 Created: ${file.name}`));
|
|
4262
|
+
created++;
|
|
4263
|
+
}
|
|
4264
|
+
}
|
|
4265
|
+
console.log("");
|
|
4266
|
+
if (created > 0) {
|
|
4267
|
+
console.log(chalk7.gray(" Next steps:"));
|
|
4268
|
+
console.log(chalk7.gray(" 1. Copy .env.example to .env and fill in your values"));
|
|
4269
|
+
console.log(chalk7.gray(" 2. Update the config file with your agent endpoint"));
|
|
4270
|
+
console.log(chalk7.gray(" 3. Run `agent-health doctor` to verify configuration"));
|
|
4271
|
+
console.log(chalk7.gray(" 4. Run `agent-health run -t sample-rca-001` to test\n"));
|
|
4272
|
+
}
|
|
4273
|
+
if (skipped > 0) {
|
|
4274
|
+
console.log(chalk7.yellow(` ${skipped} file(s) skipped. Use --force to overwrite.
|
|
4275
|
+
`));
|
|
4276
|
+
}
|
|
4277
|
+
});
|
|
4278
|
+
return command;
|
|
4279
|
+
}
|
|
4280
|
+
|
|
4281
|
+
// cli/commands/migrate.ts
|
|
4282
|
+
import { Command as Command8 } from "commander";
|
|
4283
|
+
import chalk8 from "chalk";
|
|
4284
|
+
import ora4 from "ora";
|
|
4285
|
+
function computeStatsFromReports(run, reports) {
|
|
4286
|
+
const reportsMap = new Map(reports.map((r) => [r.id, r]));
|
|
4287
|
+
let passed = 0;
|
|
4288
|
+
let failed = 0;
|
|
4289
|
+
let pending = 0;
|
|
4290
|
+
const total = Object.keys(run.results || {}).length;
|
|
4291
|
+
Object.values(run.results || {}).forEach((result) => {
|
|
4292
|
+
if (result.status === "pending" || result.status === "running") {
|
|
4293
|
+
pending++;
|
|
4294
|
+
return;
|
|
4295
|
+
}
|
|
4296
|
+
if (result.status === "failed" || result.status === "cancelled") {
|
|
4297
|
+
failed++;
|
|
4298
|
+
return;
|
|
4299
|
+
}
|
|
4300
|
+
if (result.status === "completed" && result.reportId) {
|
|
4301
|
+
const report = reportsMap.get(result.reportId);
|
|
4302
|
+
if (!report) {
|
|
4303
|
+
pending++;
|
|
4304
|
+
return;
|
|
4305
|
+
}
|
|
4306
|
+
if (report.metricsStatus === "pending" || report.metricsStatus === "calculating") {
|
|
4307
|
+
pending++;
|
|
4308
|
+
return;
|
|
4309
|
+
}
|
|
4310
|
+
if (report.passFailStatus === "passed") {
|
|
4311
|
+
passed++;
|
|
4312
|
+
} else {
|
|
4313
|
+
failed++;
|
|
4314
|
+
}
|
|
4315
|
+
} else {
|
|
4316
|
+
pending++;
|
|
4317
|
+
}
|
|
4318
|
+
});
|
|
4319
|
+
return { passed, failed, pending, total };
|
|
4320
|
+
}
|
|
4321
|
+
function createMigrateCommand() {
|
|
4322
|
+
const command = new Command8("migrate").description("One-time migration to add stats to existing benchmark runs").option("--dry-run", "Show what would be migrated without making changes").option("-v, --verbose", "Show detailed progress").action(async (options) => {
|
|
4323
|
+
console.log(chalk8.cyan.bold("\n Benchmark Stats Migration\n"));
|
|
4324
|
+
const config = await loadConfig();
|
|
4325
|
+
const serverResult = await ensureServer(config.server);
|
|
4326
|
+
const cleanup = createServerCleanup(serverResult, config.server.reuseExistingServer === false);
|
|
4327
|
+
try {
|
|
4328
|
+
const client = new ApiClient(serverResult.baseUrl);
|
|
4329
|
+
const spinner = ora4("Fetching benchmarks...").start();
|
|
4330
|
+
const benchmarks = await client.listBenchmarks();
|
|
4331
|
+
spinner.succeed(`Found ${benchmarks.length} benchmarks`);
|
|
4332
|
+
const migratable = benchmarks.filter(
|
|
4333
|
+
(b) => !b.id.startsWith("demo-") && (b.runs?.length ?? 0) > 0
|
|
4334
|
+
);
|
|
4335
|
+
if (migratable.length === 0) {
|
|
4336
|
+
console.log(chalk8.yellow("\n No benchmarks to migrate.\n"));
|
|
4337
|
+
console.log(chalk8.gray(" Only user-created benchmarks with runs can be migrated."));
|
|
4338
|
+
console.log(chalk8.gray(" Sample data (demo-*) already has stats computed.\n"));
|
|
4339
|
+
return;
|
|
4340
|
+
}
|
|
4341
|
+
console.log(chalk8.gray(`
|
|
4342
|
+
Migrating ${migratable.length} benchmarks with runs...
|
|
4343
|
+
`));
|
|
4344
|
+
let totalRuns = 0;
|
|
4345
|
+
let migratedRuns = 0;
|
|
4346
|
+
let skippedRuns = 0;
|
|
4347
|
+
let errors = 0;
|
|
4348
|
+
for (const benchmark of migratable) {
|
|
4349
|
+
const runs = benchmark.runs || [];
|
|
4350
|
+
totalRuns += runs.length;
|
|
4351
|
+
if (options.verbose) {
|
|
4352
|
+
console.log(chalk8.gray(` Processing: ${benchmark.name} (${runs.length} runs)`));
|
|
4353
|
+
}
|
|
4354
|
+
for (const run of runs) {
|
|
4355
|
+
if (run.stats && typeof run.stats.passed === "number") {
|
|
4356
|
+
skippedRuns++;
|
|
4357
|
+
if (options.verbose) {
|
|
4358
|
+
console.log(chalk8.gray(` \u2713 ${run.name} - already has stats`));
|
|
4359
|
+
}
|
|
4360
|
+
continue;
|
|
4361
|
+
}
|
|
4362
|
+
try {
|
|
4363
|
+
const reportsRes = await fetch(
|
|
4364
|
+
`${serverResult.baseUrl}/api/storage/runs/by-benchmark-run/${benchmark.id}/${run.id}`
|
|
4365
|
+
);
|
|
4366
|
+
if (!reportsRes.ok) {
|
|
4367
|
+
throw new Error(`Failed to fetch reports: ${reportsRes.status}`);
|
|
4368
|
+
}
|
|
4369
|
+
const { runs: reports } = await reportsRes.json();
|
|
4370
|
+
const stats = computeStatsFromReports(run, reports || []);
|
|
4371
|
+
if (options.verbose) {
|
|
4372
|
+
console.log(chalk8.gray(
|
|
4373
|
+
` \u2192 ${run.name}: passed=${stats.passed}, failed=${stats.failed}, pending=${stats.pending}`
|
|
4374
|
+
));
|
|
4375
|
+
}
|
|
4376
|
+
if (!options.dryRun) {
|
|
4377
|
+
const updateRes = await fetch(
|
|
4378
|
+
`${serverResult.baseUrl}/api/storage/benchmarks/${benchmark.id}/runs/${run.id}/stats`,
|
|
4379
|
+
{
|
|
4380
|
+
method: "PATCH",
|
|
4381
|
+
headers: { "Content-Type": "application/json" },
|
|
4382
|
+
body: JSON.stringify(stats)
|
|
4383
|
+
}
|
|
4384
|
+
);
|
|
4385
|
+
if (!updateRes.ok) {
|
|
4386
|
+
const errorBody = await updateRes.text();
|
|
4387
|
+
throw new Error(`Failed to update stats: ${errorBody}`);
|
|
4388
|
+
}
|
|
4389
|
+
}
|
|
4390
|
+
migratedRuns++;
|
|
4391
|
+
} catch (error) {
|
|
4392
|
+
errors++;
|
|
4393
|
+
const msg = error instanceof Error ? error.message : "Unknown error";
|
|
4394
|
+
if (options.verbose) {
|
|
4395
|
+
console.log(chalk8.red(` \u2717 ${run.name} - ${msg}`));
|
|
4396
|
+
}
|
|
4397
|
+
}
|
|
4398
|
+
}
|
|
4399
|
+
console.log(
|
|
4400
|
+
options.dryRun ? chalk8.blue(` [DRY RUN] ${benchmark.name} - ${runs.length} runs would be processed`) : chalk8.green(` \u2713 ${benchmark.name} - ${runs.length} runs`)
|
|
4401
|
+
);
|
|
4402
|
+
}
|
|
4403
|
+
console.log(chalk8.bold("\n Migration Summary\n"));
|
|
4404
|
+
console.log(chalk8.gray(` Total runs: ${totalRuns}`));
|
|
4405
|
+
console.log(chalk8.green(` Migrated: ${migratedRuns}`));
|
|
4406
|
+
console.log(chalk8.yellow(` Already done: ${skippedRuns}`));
|
|
4407
|
+
if (errors > 0) {
|
|
4408
|
+
console.log(chalk8.red(` Errors: ${errors}`));
|
|
4409
|
+
}
|
|
4410
|
+
if (options.dryRun) {
|
|
4411
|
+
console.log(chalk8.blue("\n This was a dry run. No changes were made."));
|
|
4412
|
+
console.log(chalk8.blue(" Run without --dry-run to apply changes.\n"));
|
|
4413
|
+
} else {
|
|
4414
|
+
console.log(chalk8.green("\n Migration complete!\n"));
|
|
4415
|
+
}
|
|
4416
|
+
} catch (error) {
|
|
4417
|
+
const msg = error instanceof Error ? error.message : "Unknown error";
|
|
4418
|
+
console.error(chalk8.red(`
|
|
4419
|
+
Error: ${msg}
|
|
4420
|
+
`));
|
|
4421
|
+
process.exit(1);
|
|
4422
|
+
} finally {
|
|
4423
|
+
cleanup();
|
|
4424
|
+
}
|
|
4425
|
+
});
|
|
4426
|
+
return command;
|
|
4427
|
+
}
|
|
4428
|
+
|
|
4429
|
+
// cli/index.ts
|
|
4430
|
+
var __filename3 = fileURLToPath3(import.meta.url);
|
|
4431
|
+
var __dirname3 = dirname3(__filename3);
|
|
4432
|
+
var packageJsonPath2 = join4(__dirname3, "..", "..", "package.json");
|
|
4433
|
+
var version = "0.1.0";
|
|
4434
|
+
try {
|
|
4435
|
+
const packageJson = JSON.parse(readFileSync3(packageJsonPath2, "utf-8"));
|
|
4436
|
+
version = packageJson.version;
|
|
4437
|
+
} catch {
|
|
4438
|
+
}
|
|
4439
|
+
function loadEnvFile(envPath) {
|
|
4440
|
+
const absolutePath = resolve4(process.cwd(), envPath);
|
|
4441
|
+
if (!existsSync5(absolutePath)) {
|
|
4442
|
+
console.error(chalk9.red(`
|
|
4443
|
+
Error: Environment file not found: ${absolutePath}
|
|
4444
|
+
`));
|
|
4445
|
+
process.exit(1);
|
|
4446
|
+
}
|
|
4447
|
+
const result = loadDotenv({ path: absolutePath });
|
|
4448
|
+
if (result.error) {
|
|
4449
|
+
console.error(chalk9.red(`
|
|
4450
|
+
Error loading environment file: ${result.error.message}
|
|
4451
|
+
`));
|
|
4452
|
+
process.exit(1);
|
|
4453
|
+
}
|
|
4454
|
+
console.log(chalk9.gray(` Loaded environment from: ${envPath}`));
|
|
4455
|
+
}
|
|
4456
|
+
var defaultEnvPath = resolve4(process.cwd(), ".env");
|
|
4457
|
+
if (existsSync5(defaultEnvPath)) {
|
|
4458
|
+
loadDotenv({ path: defaultEnvPath });
|
|
4459
|
+
}
|
|
4460
|
+
var program = new Command9();
|
|
4461
|
+
program.name("agent-health").description("Agent Health Evaluation Framework - Evaluate and monitor AI agent performance").version(version).enablePositionalOptions().passThroughOptions();
|
|
4462
|
+
program.option("-p, --port <number>", "Server port", "4001").option("-e, --env-file <path>", "Load environment variables from file (e.g., .env)").option("--no-browser", "Do not open browser automatically");
|
|
4463
|
+
program.action(async (options) => {
|
|
4464
|
+
console.log(chalk9.cyan.bold(`
|
|
4465
|
+
Agent Health v${version} - AI Agent Evaluation Framework
|
|
4466
|
+
`));
|
|
4467
|
+
console.log(chalk9.gray(` Working directory: ${process.cwd()}`));
|
|
4468
|
+
console.log(chalk9.gray(` Package directory: ${__dirname3}`));
|
|
4469
|
+
if (options.envFile) {
|
|
4470
|
+
loadEnvFile(options.envFile);
|
|
4471
|
+
} else if (existsSync5(defaultEnvPath)) {
|
|
4472
|
+
console.log(chalk9.gray(" Auto-loaded .env from current directory"));
|
|
4473
|
+
}
|
|
4474
|
+
const port = parseInt(options.port, 10);
|
|
4475
|
+
const spinner = ora5("Starting server...").start();
|
|
4476
|
+
try {
|
|
4477
|
+
await startServer({ port });
|
|
4478
|
+
spinner.succeed("Server started");
|
|
4479
|
+
console.log(chalk9.gray("\n Configuration:"));
|
|
4480
|
+
console.log(chalk9.gray(` Storage: Sample data (configure OpenSearch for persistence)`));
|
|
4481
|
+
console.log(chalk9.gray(` Agent: Select in UI (Demo Agent for mock, real agents require endpoints)`));
|
|
4482
|
+
console.log(chalk9.gray(` Judge: Select in UI (Demo Judge for mock, Bedrock requires AWS creds)
|
|
4483
|
+
`));
|
|
4484
|
+
const url = `http://localhost:${port}`;
|
|
4485
|
+
console.log(chalk9.green(` Server running at ${chalk9.bold(url)}
|
|
4486
|
+
`));
|
|
4487
|
+
if (options.browser !== false) {
|
|
4488
|
+
console.log(chalk9.gray(" Opening browser..."));
|
|
4489
|
+
await open(url);
|
|
4490
|
+
}
|
|
4491
|
+
console.log(chalk9.gray(" Press Ctrl+C to stop\n"));
|
|
4492
|
+
} catch (error) {
|
|
4493
|
+
spinner.fail("Failed to start server");
|
|
4494
|
+
console.error(chalk9.red(`
|
|
4495
|
+
Error: ${error instanceof Error ? error.message : error}
|
|
4496
|
+
`));
|
|
4497
|
+
process.exit(1);
|
|
4498
|
+
}
|
|
4499
|
+
});
|
|
4500
|
+
program.addCommand(createListCommand());
|
|
4501
|
+
program.addCommand(createRunCommand());
|
|
4502
|
+
program.addCommand(createBenchmarkCommand());
|
|
4503
|
+
program.addCommand(createExportCommand());
|
|
4504
|
+
program.addCommand(createReportCommand());
|
|
4505
|
+
program.addCommand(createDoctorCommand());
|
|
4506
|
+
program.addCommand(createInitCommand());
|
|
4507
|
+
program.addCommand(createMigrateCommand());
|
|
4508
|
+
program.command("serve").description("Start the Agent Health server (same as default action)").option("-p, --port <number>", "Server port", "4001").option("--no-browser", "Do not open browser automatically").action(async (options) => {
|
|
4509
|
+
console.log(chalk9.cyan.bold(`
|
|
4510
|
+
Agent Health v${version} - AI Agent Evaluation Framework
|
|
4511
|
+
`));
|
|
4512
|
+
const port = parseInt(options.port, 10);
|
|
4513
|
+
const spinner = ora5("Starting server...").start();
|
|
4514
|
+
try {
|
|
4515
|
+
await startServer({ port });
|
|
4516
|
+
spinner.succeed("Server started");
|
|
4517
|
+
const url = `http://localhost:${port}`;
|
|
4518
|
+
console.log(chalk9.green(` Server running at ${chalk9.bold(url)}
|
|
4519
|
+
`));
|
|
4520
|
+
if (options.browser !== false) {
|
|
4521
|
+
console.log(chalk9.gray(" Opening browser..."));
|
|
4522
|
+
await open(url);
|
|
4523
|
+
}
|
|
4524
|
+
console.log(chalk9.gray(" Press Ctrl+C to stop\n"));
|
|
4525
|
+
} catch (error) {
|
|
4526
|
+
spinner.fail("Failed to start server");
|
|
4527
|
+
console.error(chalk9.red(`
|
|
4528
|
+
Error: ${error instanceof Error ? error.message : error}
|
|
4529
|
+
`));
|
|
4530
|
+
process.exit(1);
|
|
4531
|
+
}
|
|
4532
|
+
});
|
|
4533
|
+
program.on("command:*", (operands) => {
|
|
4534
|
+
const unknownCommand = operands[0];
|
|
4535
|
+
const availableCommands = program.commands.map((cmd) => cmd.name());
|
|
4536
|
+
console.error(chalk9.red(`
|
|
4537
|
+
Error: Unknown command '${unknownCommand}'`));
|
|
4538
|
+
console.log("");
|
|
4539
|
+
console.log(chalk9.cyan(" Available commands:"));
|
|
4540
|
+
for (const cmd of availableCommands) {
|
|
4541
|
+
console.log(chalk9.gray(` - ${cmd}`));
|
|
4542
|
+
}
|
|
4543
|
+
console.log("");
|
|
4544
|
+
console.log(chalk9.gray(` Run ${chalk9.cyan("agent-health --help")} for usage information.
|
|
4545
|
+
`));
|
|
4546
|
+
process.exitCode = 1;
|
|
4547
|
+
});
|
|
4548
|
+
program.parse();
|