@databricks/appkit 0.74.1 → 0.75.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +3 -0
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +4 -4
- package/dist/beta.js +3 -3
- package/dist/cli/commands/agent/eval.js +78 -6
- package/dist/cli/commands/agent/eval.js.map +1 -1
- package/dist/cli/commands/registry/add.js +1 -1
- package/dist/cli/commands/registry/config-writer.js +1 -1
- package/dist/cli/index.js +4 -1
- package/dist/cli/index.js.map +1 -1
- package/dist/evals/discover.d.ts +8 -1
- package/dist/evals/discover.d.ts.map +1 -1
- package/dist/evals/discover.js +11 -1
- package/dist/evals/discover.js.map +1 -1
- package/dist/evals/index.d.ts +3 -3
- package/dist/evals/index.js +2 -2
- package/dist/evals/run-evals.d.ts +9 -2
- package/dist/evals/run-evals.d.ts.map +1 -1
- package/dist/evals/run-evals.js +13 -2
- package/dist/evals/run-evals.js.map +1 -1
- package/dist/evals/types.d.ts +40 -2
- package/dist/evals/types.d.ts.map +1 -1
- package/dist/plugins/server/index.js +2 -2
- package/dist/plugins/server/index.js.map +1 -1
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js +3 -3
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js.map +1 -1
- package/dist/plugins/server/static-server.js +3 -3
- package/dist/plugins/server/static-server.js.map +1 -1
- package/dist/plugins/server/utils.js +3 -3
- package/dist/plugins/server/utils.js.map +1 -1
- package/dist/plugins/server/vite-dev-server.js +4 -4
- package/dist/plugins/server/vite-dev-server.js.map +1 -1
- package/dist/registry/manifest-loader.d.ts +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +33 -33
- package/dist/type-generator/database/generate.js +3 -3
- package/dist/type-generator/database/generate.js.map +1 -1
- package/dist/type-generator/migration.js +2 -2
- package/dist/type-generator/migration.js.map +1 -1
- package/dist/type-generator/serving/server-file-extractor.js +3 -3
- package/dist/type-generator/serving/server-file-extractor.js.map +1 -1
- package/docs/api/appkit/Function.findRootEvalConfig.md +18 -0
- package/docs/api/appkit/Function.loadRootEvalConfig.md +18 -0
- package/docs/api/appkit/Interface.EvalWebServer.md +47 -0
- package/docs/api/appkit.md +3 -0
- package/llms.txt +3 -0
- package/package.json +1 -1
- package/sbom.cdx.json +1 -1
package/CLAUDE.md
CHANGED
|
@@ -110,6 +110,7 @@ npx @databricks/appkit docs <query>
|
|
|
110
110
|
- [Function: evalGlyph()](./docs/api/appkit/Function.evalGlyph.md): Status glyph for a single eval result.
|
|
111
111
|
- [Function: executeFromRegistry()](./docs/api/appkit/Function.executeFromRegistry.md): Validates tool-call arguments against the entry's schema and invokes its
|
|
112
112
|
- [Function: extractServingEndpoints()](./docs/api/appkit/Function.extractServingEndpoints.md): Extract serving endpoint config from a server file by AST-parsing it.
|
|
113
|
+
- [Function: findRootEvalConfig()](./docs/api/appkit/Function.findRootEvalConfig.md): Path to the root evals.config.ts (from defineEvalConfig) at
|
|
113
114
|
- [Function: findServerFile()](./docs/api/appkit/Function.findServerFile.md): Find the server entry file by checking candidate paths in order.
|
|
114
115
|
- [Function: fk()](./docs/api/appkit/Function.fk.md): Declare foreign-key to another column.
|
|
115
116
|
- [Function: formatEvalDetail()](./docs/api/appkit/Function.formatEvalDetail.md): Indented detail lines for a failing eval (error + failing assertions).
|
|
@@ -140,6 +141,7 @@ npx @databricks/appkit docs <query>
|
|
|
140
141
|
- [Function: jsonb()](./docs/api/appkit/Function.jsonb.md): Returns
|
|
141
142
|
- [Function: loadAgentFromFile()](./docs/api/appkit/Function.loadAgentFromFile.md): Loads a single markdown agent file and resolves its frontmatter against
|
|
142
143
|
- [Function: loadAgentsFromDir()](./docs/api/appkit/Function.loadAgentsFromDir.md): Scans a directory for one subdirectory per agent, each containing
|
|
144
|
+
- [Function: loadRootEvalConfig()](./docs/api/appkit/Function.loadRootEvalConfig.md): Load the root evals.config.ts under rootDir (the project root), or return
|
|
143
145
|
- [Function: matches()](./docs/api/appkit/Function.matches.md): Passes when the value matches pattern.
|
|
144
146
|
- [Function: mcpServer()](./docs/api/appkit/Function.mcpServer.md): Factory for declaring a custom MCP server tool.
|
|
145
147
|
- [Function: normalizeHost()](./docs/api/appkit/Function.normalizeHost.md): Ensure the host has a scheme (Databricks env often lacks https://).
|
|
@@ -189,6 +191,7 @@ npx @databricks/appkit docs <query>
|
|
|
189
191
|
- [Interface: EvalResult](./docs/api/appkit/Interface.EvalResult.md): The outcome of running one eval.
|
|
190
192
|
- [Interface: EvalRunSummary](./docs/api/appkit/Interface.EvalRunSummary.md): Properties
|
|
191
193
|
- [Interface: EvalSummary](./docs/api/appkit/Interface.EvalSummary.md): Properties
|
|
194
|
+
- [Interface: EvalWebServer](./docs/api/appkit/Interface.EvalWebServer.md): Auto-start config for the app under test, à la Playwright's webServer. When
|
|
192
195
|
- [Interface: FilePolicyUser](./docs/api/appkit/Interface.FilePolicyUser.md): Minimal user identity passed to the policy function.
|
|
193
196
|
- [Interface: FileResource](./docs/api/appkit/Interface.FileResource.md): Describes the file or directory being acted upon.
|
|
194
197
|
- [Interface: FunctionTool](./docs/api/appkit/Interface.FunctionTool.md): Properties
|
package/dist/appkit/package.js
CHANGED
package/dist/beta.d.ts
CHANGED
|
@@ -18,16 +18,16 @@ import { fk } from "./database/schema-builder/fk.js";
|
|
|
18
18
|
import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth, resolveWorkspaceClient } from "./connectors/mlflow/auth.js";
|
|
19
19
|
import { MlflowClient, PostResult, normalizeHost } from "./connectors/mlflow/client.js";
|
|
20
20
|
import { DatasetRow, ReadEvalDatasetOptions, readEvalDataset, userTurns } from "./evals/dataset.js";
|
|
21
|
-
import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext } from "./evals/types.js";
|
|
21
|
+
import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, EvalWebServer, MatchResult, Matcher, Severity, TestContext } from "./evals/types.js";
|
|
22
22
|
import { defineEval, defineEvalConfig } from "./evals/define-eval.js";
|
|
23
|
-
import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles } from "./evals/discover.js";
|
|
23
|
+
import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./evals/discover.js";
|
|
24
24
|
import { HttpDriverOptions, createHttpDriver } from "./evals/http-driver.js";
|
|
25
25
|
import { JudgeConfig, JudgeScore, configureJudge, isJudgeConfigured } from "./evals/judge.js";
|
|
26
26
|
import { equals, includes, matches } from "./evals/matchers.js";
|
|
27
27
|
import { Assessment, ReportOutcome, buildAssessments, reportToMlflow } from "./evals/mlflow-report.js";
|
|
28
28
|
import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./evals/report.js";
|
|
29
29
|
import { RunEvalOptions, runEval } from "./evals/run-eval.js";
|
|
30
|
-
import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries } from "./evals/run-evals.js";
|
|
30
|
+
import { EvalProgress, EvalRunSummary, RunEvalsOptions, loadRootEvalConfig, runEvalsInDir, runWithRetries } from "./evals/run-evals.js";
|
|
31
31
|
import "./evals/index.js";
|
|
32
32
|
import { agentIdFromMarkdownPath, loadAgentFromFile, loadAgentsFromDir } from "./core/agent/load-agents.js";
|
|
33
33
|
import { agents } from "./plugins/agents/agents.js";
|
|
@@ -40,4 +40,4 @@ import { DatabaseApiConfig, DatabaseApiWriteOperation, DatabaseApiWritesConfig,
|
|
|
40
40
|
import { database } from "./plugins/database/database.js";
|
|
41
41
|
import "./plugins/database/index.js";
|
|
42
42
|
import "./plugins/beta-exports.generated.js";
|
|
43
|
-
export { type AgentAdapter, type AgentDefinition, type AgentEvent, type AgentInput, type AgentRunContext, type AgentTool, type AgentToolDefinition, type AgentTools, type AgentToolsFn, type AgentsPluginConfig, AppKitMcpClient, AssertionHandle, AssertionResult, Assessment, type AutoInheritToolsConfig, type BaseSystemPromptOption, CustomJudgeSpec, type DatabaseApiConfig, type DatabaseApiWriteOperation, type DatabaseApiWritesConfig, type DatabaseExports, DatabricksAdapter, DatabricksAuth, DatasetRow, DiscoveredEval, DiscoveredEvalConfig, DriveResult, type EntityHooks, type EntityMutationHooks, EvalDefinition, EvalDriver, EvalProgress, EvalResult, EvalRunSummary, EvalSummary, type FunctionTool, type GenerationParams, type HookApp, type HookContext, type HostedSupervisorTool, type HostedTool, HttpDriverOptions, type IAiSearchConfig, type IDatabaseConfig, type IndexConfig, JudgeConfig, JudgeScore, MatchResult, Matcher, type McpConnectAllResult, type Message, MlflowClient, type PluginToolkitProvider, type Plugins, PostResult, type PromptContext, ReadEvalDatasetOptions, type ReadSerializer, type ReadSerializerContext, type RegisteredAgent, ReportOutcome, type RerankerConfig, ResolveDatabricksAuthOptions, type ResolvedToolEntry, type RunAgentInput, type RunAgentResult, RunEvalOptions, RunEvalsOptions, SUPERVISOR_EXTENSION_KEY, type Schema, type SearchFilters, type SearchRequest, type SearchResponse, type SearchResult, Severity, SupervisorApiAdapter, type SupervisorApiAdapterOptions, type SupervisorExtension, type SupervisorTool, TestContext, type Thread, type ThreadStore, type ToolAnnotations, type ToolConfig, type ToolEntry, type ToolProvider, type ToolRegistry, type ToolkitEntry, type ToolkitOptions, type TransactionClient, type WorkspaceClientLike, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineEvalConfig, defineSchema, defineTool, discoverEvalConfigs, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, readEvalDataset, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, resolveWorkspaceClient, runAgent, runEval, runEvalsInDir, runWithRetries, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, userTurns, uuid, varchar };
|
|
43
|
+
export { type AgentAdapter, type AgentDefinition, type AgentEvent, type AgentInput, type AgentRunContext, type AgentTool, type AgentToolDefinition, type AgentTools, type AgentToolsFn, type AgentsPluginConfig, AppKitMcpClient, AssertionHandle, AssertionResult, Assessment, type AutoInheritToolsConfig, type BaseSystemPromptOption, CustomJudgeSpec, type DatabaseApiConfig, type DatabaseApiWriteOperation, type DatabaseApiWritesConfig, type DatabaseExports, DatabricksAdapter, DatabricksAuth, DatasetRow, DiscoveredEval, DiscoveredEvalConfig, DriveResult, type EntityHooks, type EntityMutationHooks, EvalDefinition, EvalDriver, EvalProgress, EvalResult, EvalRunSummary, EvalSummary, EvalWebServer, type FunctionTool, type GenerationParams, type HookApp, type HookContext, type HostedSupervisorTool, type HostedTool, HttpDriverOptions, type IAiSearchConfig, type IDatabaseConfig, type IndexConfig, JudgeConfig, JudgeScore, MatchResult, Matcher, type McpConnectAllResult, type Message, MlflowClient, type PluginToolkitProvider, type Plugins, PostResult, type PromptContext, ReadEvalDatasetOptions, type ReadSerializer, type ReadSerializerContext, type RegisteredAgent, ReportOutcome, type RerankerConfig, ResolveDatabricksAuthOptions, type ResolvedToolEntry, type RunAgentInput, type RunAgentResult, RunEvalOptions, RunEvalsOptions, SUPERVISOR_EXTENSION_KEY, type Schema, type SearchFilters, type SearchRequest, type SearchResponse, type SearchResult, Severity, SupervisorApiAdapter, type SupervisorApiAdapterOptions, type SupervisorExtension, type SupervisorTool, TestContext, type Thread, type ThreadStore, type ToolAnnotations, type ToolConfig, type ToolEntry, type ToolProvider, type ToolRegistry, type ToolkitEntry, type ToolkitOptions, type TransactionClient, type WorkspaceClientLike, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineEvalConfig, defineSchema, defineTool, discoverEvalConfigs, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, findRootEvalConfig, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, loadRootEvalConfig, matches, mcpServer, normalizeHost, parseTextToolCalls, readEvalDataset, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, resolveWorkspaceClient, runAgent, runEval, runEvalsInDir, runWithRetries, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, userTurns, uuid, varchar };
|
package/dist/beta.js
CHANGED
|
@@ -17,14 +17,14 @@ import "./core/agent/tools/index.js";
|
|
|
17
17
|
import "./database/schema-builder/index.js";
|
|
18
18
|
import { readEvalDataset, userTurns } from "./evals/dataset.js";
|
|
19
19
|
import { defineEval, defineEvalConfig } from "./evals/define-eval.js";
|
|
20
|
-
import { discoverEvalConfigs, discoverEvalFiles } from "./evals/discover.js";
|
|
20
|
+
import { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./evals/discover.js";
|
|
21
21
|
import { createHttpDriver } from "./evals/http-driver.js";
|
|
22
22
|
import { configureJudge, isJudgeConfigured } from "./evals/judge.js";
|
|
23
23
|
import { equals, includes, matches } from "./evals/matchers.js";
|
|
24
24
|
import { buildAssessments, reportToMlflow } from "./evals/mlflow-report.js";
|
|
25
25
|
import { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./evals/report.js";
|
|
26
26
|
import { runEval } from "./evals/run-eval.js";
|
|
27
|
-
import { runEvalsInDir, runWithRetries } from "./evals/run-evals.js";
|
|
27
|
+
import { loadRootEvalConfig, runEvalsInDir, runWithRetries } from "./evals/run-evals.js";
|
|
28
28
|
import "./evals/index.js";
|
|
29
29
|
import { agentIdFromMarkdownPath, loadAgentFromFile, loadAgentsFromDir } from "./core/agent/load-agents.js";
|
|
30
30
|
import { agents } from "./plugins/agents/agents.js";
|
|
@@ -33,4 +33,4 @@ import { aiSearch } from "./plugins/ai-search/ai-search.js";
|
|
|
33
33
|
import { database } from "./plugins/database/database.js";
|
|
34
34
|
import "./plugins/beta-exports.generated.js";
|
|
35
35
|
|
|
36
|
-
export { AppKitMcpClient, DatabricksAdapter, MlflowClient, SUPERVISOR_EXTENSION_KEY, SupervisorApiAdapter, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineEvalConfig, defineSchema, defineTool, discoverEvalConfigs, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, readEvalDataset, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, resolveWorkspaceClient, runAgent, runEval, runEvalsInDir, runWithRetries, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, userTurns, uuid, varchar };
|
|
36
|
+
export { AppKitMcpClient, DatabricksAdapter, MlflowClient, SUPERVISOR_EXTENSION_KEY, SupervisorApiAdapter, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineEvalConfig, defineSchema, defineTool, discoverEvalConfigs, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, findRootEvalConfig, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, loadRootEvalConfig, matches, mcpServer, normalizeHost, parseTextToolCalls, readEvalDataset, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, resolveWorkspaceClient, runAgent, runEval, runEvalsInDir, runWithRetries, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, userTurns, uuid, varchar };
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import fs from "node:fs";
|
|
2
2
|
import { Command, Option } from "commander";
|
|
3
|
+
import { spawn } from "node:child_process";
|
|
3
4
|
|
|
4
5
|
//#region src/cli/commands/agent/eval.ts
|
|
5
6
|
/**
|
|
@@ -40,6 +41,69 @@ function positiveInt(raw) {
|
|
|
40
41
|
const n = raw ? Number.parseInt(raw, 10) : NaN;
|
|
41
42
|
return n > 0 ? n : void 0;
|
|
42
43
|
}
|
|
44
|
+
/** True when `url` answers with any HTTP response (a 404 still proves it's up). */
|
|
45
|
+
async function isServerUp(url) {
|
|
46
|
+
try {
|
|
47
|
+
await fetch(url, { signal: AbortSignal.timeout(2e3) });
|
|
48
|
+
return true;
|
|
49
|
+
} catch {
|
|
50
|
+
return false;
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* Start the app under test per a root config's `webServer`, à la Playwright:
|
|
55
|
+
* reuse a server already answering at `url` (unless `reuseExisting: false`),
|
|
56
|
+
* else spawn `command`, poll `url` until it answers or `timeoutMs` elapses.
|
|
57
|
+
* Returns a `stop()` that kills the spawned process group (a no-op when the
|
|
58
|
+
* server was reused). Under a machine reporter (`machine`) our logs and the
|
|
59
|
+
* child's stdout are both routed to stderr, so the report stream on stdout
|
|
60
|
+
* stays clean. Throws if the server never comes up.
|
|
61
|
+
*/
|
|
62
|
+
async function startWebServer(webServer, baseUrl, machine) {
|
|
63
|
+
const url = webServer.url ?? baseUrl;
|
|
64
|
+
const reuse = webServer.reuseExisting !== false;
|
|
65
|
+
const noop = { stop: () => {} };
|
|
66
|
+
if (reuse && await isServerUp(url)) {
|
|
67
|
+
console.error(`Reusing server already running at ${url}`);
|
|
68
|
+
return noop;
|
|
69
|
+
}
|
|
70
|
+
console.error(`Starting web server: ${webServer.command}`);
|
|
71
|
+
const child = spawn(webServer.command, {
|
|
72
|
+
shell: true,
|
|
73
|
+
detached: true,
|
|
74
|
+
stdio: machine ? [
|
|
75
|
+
"ignore",
|
|
76
|
+
2,
|
|
77
|
+
"inherit"
|
|
78
|
+
] : "inherit"
|
|
79
|
+
});
|
|
80
|
+
let exited = false;
|
|
81
|
+
child.on("exit", () => {
|
|
82
|
+
exited = true;
|
|
83
|
+
});
|
|
84
|
+
const stop = () => {
|
|
85
|
+
if (exited || child.pid === void 0) return;
|
|
86
|
+
try {
|
|
87
|
+
process.kill(-child.pid, "SIGTERM");
|
|
88
|
+
} catch {}
|
|
89
|
+
};
|
|
90
|
+
const deadline = Date.now() + (webServer.timeoutMs ?? 6e4);
|
|
91
|
+
try {
|
|
92
|
+
while (Date.now() < deadline) {
|
|
93
|
+
if (exited) throw new Error("web server exited before becoming ready");
|
|
94
|
+
if (await isServerUp(url)) {
|
|
95
|
+
console.error(`Web server ready at ${url}`);
|
|
96
|
+
return { stop };
|
|
97
|
+
}
|
|
98
|
+
await new Promise((r) => setTimeout(r, 500));
|
|
99
|
+
}
|
|
100
|
+
} catch (err) {
|
|
101
|
+
stop();
|
|
102
|
+
throw err;
|
|
103
|
+
}
|
|
104
|
+
stop();
|
|
105
|
+
throw new Error(`web server did not respond at ${url} within ${webServer.timeoutMs ?? 6e4}ms`);
|
|
106
|
+
}
|
|
43
107
|
/**
|
|
44
108
|
* Native MLflow "Evaluation run" config — only when creds + an experiment are
|
|
45
109
|
* all present (traces live in the app; the run + scores are driven from here).
|
|
@@ -103,6 +167,9 @@ function printMlflowOutcome(mlflow, info) {
|
|
|
103
167
|
}
|
|
104
168
|
async function runAgentEval(filter, opts) {
|
|
105
169
|
const runner = await loadRunner();
|
|
170
|
+
const rootDir = opts.root ?? process.cwd();
|
|
171
|
+
const config = await runner.loadRootEvalConfig(rootDir) ?? {};
|
|
172
|
+
const baseUrl = opts.url ?? config.baseUrl ?? "http://localhost:8000";
|
|
106
173
|
const credentials = {
|
|
107
174
|
profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,
|
|
108
175
|
host: opts.databricksHost ?? process.env.DATABRICKS_HOST,
|
|
@@ -111,7 +178,8 @@ async function runAgentEval(filter, opts) {
|
|
|
111
178
|
const auth = await runner.resolveDatabricksAuth(credentials) ?? {};
|
|
112
179
|
const warehouseId = opts.warehouseId ?? process.env.DATABRICKS_WAREHOUSE_ID;
|
|
113
180
|
const workspaceClient = runner.resolveWorkspaceClient(credentials);
|
|
114
|
-
const
|
|
181
|
+
const concurrency = positiveInt(opts.concurrency) ?? config.maxConcurrency;
|
|
182
|
+
const timeoutMs = positiveInt(opts.timeout) ?? config.timeoutMs;
|
|
115
183
|
const retries = positiveInt(opts.retries);
|
|
116
184
|
let minPassRate;
|
|
117
185
|
try {
|
|
@@ -127,28 +195,32 @@ async function runAgentEval(filter, opts) {
|
|
|
127
195
|
if (machine) console.error(msg);
|
|
128
196
|
else console.log(msg);
|
|
129
197
|
};
|
|
198
|
+
let server;
|
|
130
199
|
let summary;
|
|
131
200
|
try {
|
|
201
|
+
server = config.webServer ? await startWebServer(config.webServer, baseUrl, machine) : void 0;
|
|
132
202
|
summary = await runner.runEvalsInDir({
|
|
133
|
-
rootDir
|
|
134
|
-
baseUrl
|
|
203
|
+
rootDir,
|
|
204
|
+
baseUrl,
|
|
135
205
|
filter,
|
|
136
206
|
tags: opts.tag,
|
|
137
207
|
strict: opts.strict,
|
|
138
208
|
headers: opts.header ? parseHeaders(opts.header) : void 0,
|
|
139
|
-
concurrency
|
|
209
|
+
concurrency,
|
|
140
210
|
mlflow: resolveMlflow(opts, auth),
|
|
141
211
|
judge: resolveJudge(opts, auth),
|
|
142
212
|
workspaceClient,
|
|
143
213
|
warehouseId,
|
|
144
214
|
timeoutMs,
|
|
145
215
|
retries,
|
|
146
|
-
onEvent: makeProgressReporter(runner,
|
|
216
|
+
onEvent: makeProgressReporter(runner, baseUrl, machine, info)
|
|
147
217
|
});
|
|
148
218
|
} catch (err) {
|
|
149
219
|
console.error(`\nEval run failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
150
220
|
process.exitCode = 1;
|
|
151
221
|
return;
|
|
222
|
+
} finally {
|
|
223
|
+
server?.stop();
|
|
152
224
|
}
|
|
153
225
|
info(`\n${runner.formatSummaryLine(summary.results)}`);
|
|
154
226
|
if (summary.mlflow) printMlflowOutcome(summary.mlflow, info);
|
|
@@ -173,7 +245,7 @@ async function runAgentEval(filter, opts) {
|
|
|
173
245
|
if (!ok) process.exitCode = 1;
|
|
174
246
|
} else if (!stats.allPassed) process.exitCode = 1;
|
|
175
247
|
}
|
|
176
|
-
const agentEvalCommand = new Command("eval").description("Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app").argument("[filter]", "Only run evals whose <agent>/<id> contains this substring (or an exact agent id)").option("--url <url>", "Base URL of the
|
|
248
|
+
const agentEvalCommand = new Command("eval").description("Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app").argument("[filter]", "Only run evals whose <agent>/<id> contains this substring (or an exact agent id)").option("--url <url>", "Base URL of the app to drive (default: evals.config.ts baseUrl, else http://localhost:8000)").option("--strict", "Fail on soft-assertion misses too", false).option("--concurrency <n>", "Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)").option("--root <dir>", "Project root containing server/agents/ (default: cwd)").option("--header <header...>", "Extra request header as 'Key: value' (repeatable)").option("--tag <tag...>", "Only run evals tagged with one of these tags (repeatable)").option("--profile <name>", "Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)").option("--databricks-host <host>", "Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)").option("--databricks-token <token>", "Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)").option("--experiment <id>", "MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)").option("--warehouse-id <id>", "SQL warehouse id for reading managed eval datasets and writing assessments to UC-backed experiments (default: DATABRICKS_WAREHOUSE_ID, or MLFLOW_TRACING_SQL_WAREHOUSE_ID for assessments)").option("--judge-model <endpoint>", "Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)").option("--timeout <ms>", "Default per-eval timeout in ms (a per-eval timeoutMs overrides it)").option("--retries <n>", "Re-run an eval up to N times when it fails on an infra error (turn/timeout); assertion failures are not retried").option("--min-pass-rate <rate>", "Gate on aggregate pass rate (0..1) instead of requiring every eval to pass; exit 1 when below").addOption(new Option("--reporter <format>", "Report format: text (live console), json (dashboards), or junit (CI test reporters)").choices([
|
|
177
249
|
"text",
|
|
178
250
|
"json",
|
|
179
251
|
"junit"
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"eval.js","names":[],"sources":["../../../../src/cli/commands/agent/eval.ts"],"sourcesContent":["import fs from \"node:fs\";\n\nimport { Command, Option } from \"commander\";\n\ninterface EvalRunSummary {\n results: unknown[];\n mlflow?: {\n runId: string;\n report: {\n written: number;\n skipped: number;\n failures: Array<{ traceId: string; status?: number; error?: string }>;\n };\n finish: { finished: boolean; metricsError?: string; finishError?: string };\n };\n}\n\ntype EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: unknown; index: number; total: number };\n\n/** Subset of `@databricks/appkit/beta`'s eval runner used by this command. */\ninterface EvalRunner {\n runEvalsInDir(opts: {\n rootDir?: string;\n baseUrl: string;\n filter?: string;\n tags?: string[];\n strict?: boolean;\n headers?: Record<string, string>;\n concurrency?: number;\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n sqlWarehouseId?: string;\n };\n judge?: { host: string; token: string; model: string };\n workspaceClient?: unknown;\n warehouseId?: string;\n timeoutMs?: number;\n retries?: number;\n onEvent?: (event: EvalProgress) => void;\n }): Promise<EvalRunSummary>;\n resolveDatabricksAuth(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): Promise<{ host: string; token: string } | undefined>;\n resolveWorkspaceClient(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): unknown;\n formatEvalHeadline(result: unknown): string;\n evalGlyph(result: unknown): string;\n formatEvalDetail(result: unknown): string[];\n formatSummaryLine(results: unknown[]): string;\n formatResultsJson(results: unknown[]): string;\n formatResultsJUnit(results: unknown[]): string;\n summarize(results: unknown[]): { allPassed: boolean; passRate: number };\n}\n\n/**\n * Loaded at runtime from the consuming project so this command (which ships in\n * `@databricks/shared`) doesn't take a build-time dependency on appkit. The\n * specifier is a variable so the type checker treats it as `any`.\n */\nasync function loadRunner(): Promise<EvalRunner> {\n const spec = \"@databricks/appkit/beta\";\n try {\n return (await import(spec)) as unknown as EvalRunner;\n } catch (err) {\n throw new Error(\n \"Could not load @databricks/appkit. Run `appkit agent eval` from a \" +\n \"project with @databricks/appkit installed. \" +\n `Cause: ${err instanceof Error ? err.message : String(err)}`,\n );\n }\n}\n\nfunction parseHeaders(values: string[]): Record<string, string> {\n const headers: Record<string, string> = {};\n for (const v of values) {\n const i = v.indexOf(\":\");\n if (i === -1) continue;\n headers[v.slice(0, i).trim()] = v.slice(i + 1).trim();\n }\n return headers;\n}\n\n/**\n * Parse `--min-pass-rate`: a finite number in `[0, 1]`, or `undefined` when\n * unset. Throws on blank, out-of-range, or non-numeric input so a bad gate\n * value fails fast instead of silently disabling or inverting the CI gate.\n */\nexport function parsePassRate(raw: string | undefined): number | undefined {\n if (raw === undefined) return undefined;\n const n = Number(raw);\n // Reject blank too: `Number(\"\")` is 0, which would silently disable the gate.\n if (raw.trim() === \"\" || !Number.isFinite(n) || n < 0 || n > 1) {\n throw new Error(\n `Invalid --min-pass-rate \"${raw}\" — expected a number in [0, 1]`,\n );\n }\n return n;\n}\n\n/** Parse a positive-integer CLI option; junk, zero, or negative → undefined. */\nfunction positiveInt(raw: string | undefined): number | undefined {\n const n = raw ? Number.parseInt(raw, 10) : Number.NaN;\n return n > 0 ? n : undefined;\n}\n\ninterface EvalOptions {\n url: string;\n strict?: boolean;\n root?: string;\n header?: string[];\n tag?: string[];\n profile?: string;\n databricksHost?: string;\n databricksToken?: string;\n experiment?: string;\n judgeModel?: string;\n concurrency?: number;\n warehouseId?: string;\n timeout?: string;\n retries?: string;\n minPassRate?: string;\n reporter?: \"text\" | \"json\" | \"junit\";\n output?: string;\n}\n\n/** Resolved Databricks host + bearer (either field may be absent). */\ntype Auth = { host?: string; token?: string };\n\n/**\n * Native MLflow \"Evaluation run\" config — only when creds + an experiment are\n * all present (traces live in the app; the run + scores are driven from here).\n */\nfunction resolveMlflow(opts: EvalOptions, auth: Auth) {\n const experimentId = opts.experiment ?? process.env.MLFLOW_EXPERIMENT_ID;\n if (!(auth.host && auth.token && experimentId)) return undefined;\n // UC-backed experiments need a SQL warehouse to write assessments to their\n // V4 traces. Mirror mlflow's env var, and accept the common DATABRICKS one.\n const sqlWarehouseId =\n opts.warehouseId ??\n process.env.MLFLOW_TRACING_SQL_WAREHOUSE_ID ??\n process.env.DATABRICKS_WAREHOUSE_ID;\n return {\n host: auth.host,\n token: auth.token,\n experimentId,\n ...(sqlWarehouseId ? { sqlWarehouseId } : {}),\n };\n}\n\n/** LLM-as-judge config — reuses the Databricks creds + a judge serving endpoint. */\nfunction resolveJudge(opts: EvalOptions, auth: Auth) {\n const model = opts.judgeModel ?? process.env.APPKIT_JUDGE_MODEL;\n return model && auth.host && auth.token\n ? { host: auth.host, token: auth.token, model }\n : undefined;\n}\n\n/**\n * Progress reporter: stream each eval as it runs instead of going silent. In a\n * machine reporter (json/junit) the live per-eval streaming is suppressed and\n * banners go to stderr (via `info`), keeping stdout clean for the report.\n */\nfunction makeProgressReporter(\n runner: EvalRunner,\n url: string,\n machine: boolean,\n info: (msg: string) => void,\n): (event: EvalProgress) => void {\n return (event) => {\n switch (event.type) {\n case \"discovered\":\n info(\n `Running ${event.total} eval${event.total === 1 ? \"\" : \"s\"} against ${url}\\n`,\n );\n break;\n case \"run-created\":\n info(`MLflow evaluation run: ${event.runId}\\n`);\n break;\n case \"result\": {\n if (machine) break;\n // One full line per completion — evals run concurrently, so a split\n // \"start … glyph\" prefix would interleave into garbage.\n console.log(\n `[${event.index + 1}/${event.total}] ${runner.formatEvalHeadline(event.result)}`,\n );\n for (const line of runner.formatEvalDetail(event.result)) {\n console.log(line);\n }\n break;\n }\n }\n };\n}\n\nfunction formatFailureLine(f: {\n traceId: string;\n status?: number;\n error?: string;\n}): string {\n return ` ✗ trace ${f.traceId}: ${f.status ?? \"\"} ${f.error ?? \"\"}`.trim();\n}\n\n/**\n * Print the MLflow assessment/finish outcome after a run that created one. The\n * summary line goes through `info` (stderr under a machine reporter); per-trace\n * failures and finish errors always go to stderr.\n */\nfunction printMlflowOutcome(\n mlflow: NonNullable<EvalRunSummary[\"mlflow\"]>,\n info: (msg: string) => void,\n): void {\n const { report, finish } = mlflow;\n info(\n `MLflow: ${report.written} assessment(s) written` +\n (report.skipped ? `, ${report.skipped} skipped` : \"\") +\n (report.failures.length ? `, ${report.failures.length} failed` : \"\"),\n );\n for (const f of report.failures) {\n console.error(formatFailureLine(f));\n }\n if (finish.metricsError) {\n console.error(` ⚠ metrics not logged: ${finish.metricsError}`);\n }\n if (!finish.finished) {\n console.error(\n ` ✗ run left RUNNING — failed to finish: ${finish.finishError ?? \"unknown\"}`,\n );\n }\n}\n\nasync function runAgentEval(\n filter: string | undefined,\n opts: EvalOptions,\n): Promise<void> {\n const runner = await loadRunner();\n\n // Databricks credentials shared by auth resolution and the workspace client:\n // an explicit flag/DATABRICKS_* env wins, else the SDK resolves from the CLI\n // profile.\n const credentials = {\n profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,\n host: opts.databricksHost ?? process.env.DATABRICKS_HOST,\n token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN,\n };\n\n // Resolve Databricks host + bearer the AppKit-native way: an explicit\n // host/token wins; otherwise the SDK mints an OAuth token from the CLI\n // profile — so no hand-set PAT is required.\n const auth: Auth = (await runner.resolveDatabricksAuth(credentials)) ?? {};\n\n // Managed-dataset reads: a workspace client (same profile/host/token) + a SQL\n // warehouse. Only needed by evals that declare `dataset`.\n const warehouseId = opts.warehouseId ?? process.env.DATABRICKS_WAREHOUSE_ID;\n const workspaceClient = runner.resolveWorkspaceClient(credentials);\n\n // Runner-level default per-eval timeout (ms). A per-eval `timeoutMs` wins.\n const timeoutMs = positiveInt(opts.timeout);\n\n // Extra attempts for evals that fail on an infra error (turn/timeout). Junk\n // or negative input falls back to no retries.\n const retries = positiveInt(opts.retries);\n\n // Validate up front so a bad gate value fails before the run, not after.\n let minPassRate: number | undefined;\n try {\n minPassRate = parsePassRate(opts.minPassRate);\n } catch (err) {\n console.error(err instanceof Error ? err.message : String(err));\n process.exitCode = 1;\n return;\n }\n\n // In a machine reporter (json/junit), stdout is reserved for the report (it\n // may be piped), so human-facing lines go to stderr and the per-eval live\n // streaming is suppressed. Text mode keeps its current stdout behavior.\n const reporter = opts.reporter ?? \"text\";\n const machine = reporter !== \"text\";\n const info = (msg: string): void => {\n if (machine) console.error(msg);\n else console.log(msg);\n };\n\n let summary: EvalRunSummary;\n try {\n summary = await runner.runEvalsInDir({\n rootDir: opts.root,\n baseUrl: opts.url,\n filter,\n tags: opts.tag,\n strict: opts.strict,\n headers: opts.header ? parseHeaders(opts.header) : undefined,\n concurrency: opts.concurrency,\n mlflow: resolveMlflow(opts, auth),\n judge: resolveJudge(opts, auth),\n workspaceClient,\n warehouseId,\n timeoutMs,\n retries,\n onEvent: makeProgressReporter(runner, opts.url, machine, info),\n });\n } catch (err) {\n // Setup failures (e.g. a bad --experiment for the MLflow run) reject before\n // any eval runs; surface a clean message + non-zero exit rather than an\n // unhandled promise rejection with a raw stack.\n console.error(\n `\\nEval run failed: ${err instanceof Error ? err.message : String(err)}`,\n );\n process.exitCode = 1;\n return;\n }\n\n // The final human summary always shows (stderr for machine reporters so it\n // never pollutes the report on stdout/file).\n info(`\\n${runner.formatSummaryLine(summary.results)}`);\n\n if (summary.mlflow) {\n printMlflowOutcome(summary.mlflow, info);\n } else {\n info(\n \"\\nMLflow evaluation run skipped — pass --experiment (or set\" +\n \" MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.\",\n );\n }\n\n // Machine-readable report: build the string with a pure formatter, then emit\n // it to --output <file> or stdout (kept clean of the human noise above).\n if (machine) {\n const report =\n reporter === \"json\"\n ? runner.formatResultsJson(summary.results)\n : runner.formatResultsJUnit(summary.results);\n if (opts.output) {\n try {\n fs.writeFileSync(opts.output, `${report}\\n`);\n } catch (err) {\n console.error(\n `Failed to write ${reporter} report to ${opts.output}: ${\n err instanceof Error ? err.message : String(err)\n }`,\n );\n process.exitCode = 1;\n return;\n }\n info(`Wrote ${reporter} report to ${opts.output}`);\n } else {\n process.stdout.write(`${report}\\n`);\n }\n }\n\n const stats = runner.summarize(summary.results);\n if (minPassRate !== undefined) {\n // Threshold mode: gate on the aggregate pass rate rather than requiring\n // every eval to pass.\n const ok = stats.passRate >= minPassRate;\n info(\n `Pass rate ${(stats.passRate * 100).toFixed(0)}% (threshold ${(\n minPassRate * 100\n ).toFixed(0)}%) — ${ok ? \"OK\" : \"below threshold\"}`,\n );\n if (!ok) process.exitCode = 1;\n } else if (!stats.allPassed) {\n process.exitCode = 1;\n }\n}\n\nexport const agentEvalCommand = new Command(\"eval\")\n .description(\n \"Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app\",\n )\n .argument(\n \"[filter]\",\n \"Only run evals whose <agent>/<id> contains this substring (or an exact agent id)\",\n )\n .option(\"--url <url>\", \"Base URL of the running app\", \"http://localhost:3000\")\n .option(\"--strict\", \"Fail on soft-assertion misses too\", false)\n .option(\n \"--concurrency <n>\",\n \"Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)\",\n (v) => Number.parseInt(v, 10),\n )\n .option(\n \"--root <dir>\",\n \"Project root containing server/agents/ (default: cwd)\",\n )\n .option(\n \"--header <header...>\",\n \"Extra request header as 'Key: value' (repeatable)\",\n )\n .option(\n \"--tag <tag...>\",\n \"Only run evals tagged with one of these tags (repeatable)\",\n )\n .option(\n \"--profile <name>\",\n \"Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)\",\n )\n .option(\n \"--databricks-host <host>\",\n \"Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)\",\n )\n .option(\n \"--databricks-token <token>\",\n \"Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)\",\n )\n .option(\n \"--experiment <id>\",\n \"MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)\",\n )\n .option(\n \"--warehouse-id <id>\",\n \"SQL warehouse id for reading managed eval datasets and writing assessments to UC-backed experiments (default: DATABRICKS_WAREHOUSE_ID, or MLFLOW_TRACING_SQL_WAREHOUSE_ID for assessments)\",\n )\n .option(\n \"--judge-model <endpoint>\",\n \"Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)\",\n )\n .option(\n \"--timeout <ms>\",\n \"Default per-eval timeout in ms (a per-eval timeoutMs overrides it)\",\n )\n .option(\n \"--retries <n>\",\n \"Re-run an eval up to N times when it fails on an infra error (turn/timeout); assertion failures are not retried\",\n )\n .option(\n \"--min-pass-rate <rate>\",\n \"Gate on aggregate pass rate (0..1) instead of requiring every eval to pass; exit 1 when below\",\n )\n .addOption(\n new Option(\n \"--reporter <format>\",\n \"Report format: text (live console), json (dashboards), or junit (CI test reporters)\",\n )\n .choices([\"text\", \"json\", \"junit\"])\n .default(\"text\"),\n )\n .option(\n \"--output <file>\",\n \"Write the json/junit report to this file instead of stdout (ignored for text)\",\n )\n .action(runAgentEval);\n"],"mappings":";;;;;;;;;AAsEA,eAAe,aAAkC;CAC/C,MAAM,OAAO;AACb,KAAI;AACF,SAAQ,MAAM,OAAO;UACd,KAAK;AACZ,QAAM,IAAI,MACR,yHAEY,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAC7D;;;AAIL,SAAS,aAAa,QAA0C;CAC9D,MAAM,UAAkC,EAAE;AAC1C,MAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,IAAI,EAAE,QAAQ,IAAI;AACxB,MAAI,MAAM,GAAI;AACd,UAAQ,EAAE,MAAM,GAAG,EAAE,CAAC,MAAM,IAAI,EAAE,MAAM,IAAI,EAAE,CAAC,MAAM;;AAEvD,QAAO;;;;;;;AAQT,SAAgB,cAAc,KAA6C;AACzE,KAAI,QAAQ,OAAW,QAAO;CAC9B,MAAM,IAAI,OAAO,IAAI;AAErB,KAAI,IAAI,MAAM,KAAK,MAAM,CAAC,OAAO,SAAS,EAAE,IAAI,IAAI,KAAK,IAAI,EAC3D,OAAM,IAAI,MACR,4BAA4B,IAAI,iCACjC;AAEH,QAAO;;;AAIT,SAAS,YAAY,KAA6C;CAChE,MAAM,IAAI,MAAM,OAAO,SAAS,KAAK,GAAG,GAAG;AAC3C,QAAO,IAAI,IAAI,IAAI;;;;;;AA8BrB,SAAS,cAAc,MAAmB,MAAY;CACpD,MAAM,eAAe,KAAK,cAAc,QAAQ,IAAI;AACpD,KAAI,EAAE,KAAK,QAAQ,KAAK,SAAS,cAAe,QAAO;CAGvD,MAAM,iBACJ,KAAK,eACL,QAAQ,IAAI,mCACZ,QAAQ,IAAI;AACd,QAAO;EACL,MAAM,KAAK;EACX,OAAO,KAAK;EACZ;EACA,GAAI,iBAAiB,EAAE,gBAAgB,GAAG,EAAE;EAC7C;;;AAIH,SAAS,aAAa,MAAmB,MAAY;CACnD,MAAM,QAAQ,KAAK,cAAc,QAAQ,IAAI;AAC7C,QAAO,SAAS,KAAK,QAAQ,KAAK,QAC9B;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;EAAO;EAAO,GAC7C;;;;;;;AAQN,SAAS,qBACP,QACA,KACA,SACA,MAC+B;AAC/B,SAAQ,UAAU;AAChB,UAAQ,MAAM,MAAd;GACE,KAAK;AACH,SACE,WAAW,MAAM,MAAM,OAAO,MAAM,UAAU,IAAI,KAAK,IAAI,WAAW,IAAI,IAC3E;AACD;GACF,KAAK;AACH,SAAK,0BAA0B,MAAM,MAAM,IAAI;AAC/C;GACF,KAAK;AACH,QAAI,QAAS;AAGb,YAAQ,IACN,IAAI,MAAM,QAAQ,EAAE,GAAG,MAAM,MAAM,IAAI,OAAO,mBAAmB,MAAM,OAAO,GAC/E;AACD,SAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,OAAO,CACtD,SAAQ,IAAI,KAAK;AAEnB;;;;AAMR,SAAS,kBAAkB,GAIhB;AACT,QAAO,aAAa,EAAE,QAAQ,IAAI,EAAE,UAAU,GAAG,GAAG,EAAE,SAAS,KAAK,MAAM;;;;;;;AAQ5E,SAAS,mBACP,QACA,MACM;CACN,MAAM,EAAE,QAAQ,WAAW;AAC3B,MACE,WAAW,OAAO,QAAQ,2BACvB,OAAO,UAAU,KAAK,OAAO,QAAQ,YAAY,OACjD,OAAO,SAAS,SAAS,KAAK,OAAO,SAAS,OAAO,WAAW,IACpE;AACD,MAAK,MAAM,KAAK,OAAO,SACrB,SAAQ,MAAM,kBAAkB,EAAE,CAAC;AAErC,KAAI,OAAO,aACT,SAAQ,MAAM,2BAA2B,OAAO,eAAe;AAEjE,KAAI,CAAC,OAAO,SACV,SAAQ,MACN,4CAA4C,OAAO,eAAe,YACnE;;AAIL,eAAe,aACb,QACA,MACe;CACf,MAAM,SAAS,MAAM,YAAY;CAKjC,MAAM,cAAc;EAClB,SAAS,KAAK,WAAW,QAAQ,IAAI;EACrC,MAAM,KAAK,kBAAkB,QAAQ,IAAI;EACzC,OAAO,KAAK,mBAAmB,QAAQ,IAAI;EAC5C;CAKD,MAAM,OAAc,MAAM,OAAO,sBAAsB,YAAY,IAAK,EAAE;CAI1E,MAAM,cAAc,KAAK,eAAe,QAAQ,IAAI;CACpD,MAAM,kBAAkB,OAAO,uBAAuB,YAAY;CAGlE,MAAM,YAAY,YAAY,KAAK,QAAQ;CAI3C,MAAM,UAAU,YAAY,KAAK,QAAQ;CAGzC,IAAI;AACJ,KAAI;AACF,gBAAc,cAAc,KAAK,YAAY;UACtC,KAAK;AACZ,UAAQ,MAAM,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,CAAC;AAC/D,UAAQ,WAAW;AACnB;;CAMF,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,UAAU,aAAa;CAC7B,MAAM,QAAQ,QAAsB;AAClC,MAAI,QAAS,SAAQ,MAAM,IAAI;MAC1B,SAAQ,IAAI,IAAI;;CAGvB,IAAI;AACJ,KAAI;AACF,YAAU,MAAM,OAAO,cAAc;GACnC,SAAS,KAAK;GACd,SAAS,KAAK;GACd;GACA,MAAM,KAAK;GACX,QAAQ,KAAK;GACb,SAAS,KAAK,SAAS,aAAa,KAAK,OAAO,GAAG;GACnD,aAAa,KAAK;GAClB,QAAQ,cAAc,MAAM,KAAK;GACjC,OAAO,aAAa,MAAM,KAAK;GAC/B;GACA;GACA;GACA;GACA,SAAS,qBAAqB,QAAQ,KAAK,KAAK,SAAS,KAAK;GAC/D,CAAC;UACK,KAAK;AAIZ,UAAQ,MACN,sBAAsB,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GACvE;AACD,UAAQ,WAAW;AACnB;;AAKF,MAAK,KAAK,OAAO,kBAAkB,QAAQ,QAAQ,GAAG;AAEtD,KAAI,QAAQ,OACV,oBAAmB,QAAQ,QAAQ,KAAK;KAExC,MACE,oIAED;AAKH,KAAI,SAAS;EACX,MAAM,SACJ,aAAa,SACT,OAAO,kBAAkB,QAAQ,QAAQ,GACzC,OAAO,mBAAmB,QAAQ,QAAQ;AAChD,MAAI,KAAK,QAAQ;AACf,OAAI;AACF,OAAG,cAAc,KAAK,QAAQ,GAAG,OAAO,IAAI;YACrC,KAAK;AACZ,YAAQ,MACN,mBAAmB,SAAS,aAAa,KAAK,OAAO,IACnD,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAEnD;AACD,YAAQ,WAAW;AACnB;;AAEF,QAAK,SAAS,SAAS,aAAa,KAAK,SAAS;QAElD,SAAQ,OAAO,MAAM,GAAG,OAAO,IAAI;;CAIvC,MAAM,QAAQ,OAAO,UAAU,QAAQ,QAAQ;AAC/C,KAAI,gBAAgB,QAAW;EAG7B,MAAM,KAAK,MAAM,YAAY;AAC7B,OACE,cAAc,MAAM,WAAW,KAAK,QAAQ,EAAE,CAAC,gBAC7C,cAAc,KACd,QAAQ,EAAE,CAAC,OAAO,KAAK,OAAO,oBACjC;AACD,MAAI,CAAC,GAAI,SAAQ,WAAW;YACnB,CAAC,MAAM,UAChB,SAAQ,WAAW;;AAIvB,MAAa,mBAAmB,IAAI,QAAQ,OAAO,CAChD,YACC,6EACD,CACA,SACC,YACA,mFACD,CACA,OAAO,eAAe,+BAA+B,wBAAwB,CAC7E,OAAO,YAAY,qCAAqC,MAAM,CAC9D,OACC,qBACA,0GACC,MAAM,OAAO,SAAS,GAAG,GAAG,CAC9B,CACA,OACC,gBACA,wDACD,CACA,OACC,wBACA,oDACD,CACA,OACC,kBACA,4DACD,CACA,OACC,oBACA,6FACD,CACA,OACC,4BACA,4EACD,CACA,OACC,8BACA,8EACD,CACA,OACC,qBACA,8EACD,CACA,OACC,uBACA,6LACD,CACA,OACC,4BACA,kGACD,CACA,OACC,kBACA,qEACD,CACA,OACC,iBACA,kHACD,CACA,OACC,0BACA,gGACD,CACA,UACC,IAAI,OACF,uBACA,sFACD,CACE,QAAQ;CAAC;CAAQ;CAAQ;CAAQ,CAAC,CAClC,QAAQ,OAAO,CACnB,CACA,OACC,mBACA,gFACD,CACA,OAAO,aAAa"}
|
|
1
|
+
{"version":3,"file":"eval.js","names":[],"sources":["../../../../src/cli/commands/agent/eval.ts"],"sourcesContent":["import { type ChildProcess, spawn } from \"node:child_process\";\nimport fs from \"node:fs\";\n\nimport { Command, Option } from \"commander\";\n\ninterface EvalRunSummary {\n results: unknown[];\n mlflow?: {\n runId: string;\n report: {\n written: number;\n skipped: number;\n failures: Array<{ traceId: string; status?: number; error?: string }>;\n };\n finish: { finished: boolean; metricsError?: string; finishError?: string };\n };\n}\n\ntype EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: unknown; index: number; total: number };\n\n/** Subset of `@databricks/appkit/beta`'s eval runner used by this command. */\ninterface EvalRunner {\n runEvalsInDir(opts: {\n rootDir?: string;\n baseUrl: string;\n filter?: string;\n tags?: string[];\n strict?: boolean;\n headers?: Record<string, string>;\n concurrency?: number;\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n sqlWarehouseId?: string;\n };\n judge?: { host: string; token: string; model: string };\n workspaceClient?: unknown;\n warehouseId?: string;\n timeoutMs?: number;\n retries?: number;\n onEvent?: (event: EvalProgress) => void;\n }): Promise<EvalRunSummary>;\n resolveDatabricksAuth(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): Promise<{ host: string; token: string } | undefined>;\n resolveWorkspaceClient(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): unknown;\n formatEvalHeadline(result: unknown): string;\n loadRootEvalConfig(rootDir: string): Promise<EvalConfig | undefined>;\n evalGlyph(result: unknown): string;\n formatEvalDetail(result: unknown): string[];\n formatSummaryLine(results: unknown[]): string;\n formatResultsJson(results: unknown[]): string;\n formatResultsJUnit(results: unknown[]): string;\n summarize(results: unknown[]): { allPassed: boolean; passRate: number };\n}\n\n/** Subset of `@databricks/appkit/beta`'s `EvalConfig` the CLI reads. */\ninterface EvalConfig {\n maxConcurrency?: number;\n timeoutMs?: number;\n baseUrl?: string;\n webServer?: {\n command: string;\n url?: string;\n timeoutMs?: number;\n reuseExisting?: boolean;\n };\n}\n\n/**\n * Loaded at runtime from the consuming project so this command (which ships in\n * `@databricks/shared`) doesn't take a build-time dependency on appkit. The\n * specifier is a variable so the type checker treats it as `any`.\n */\nasync function loadRunner(): Promise<EvalRunner> {\n const spec = \"@databricks/appkit/beta\";\n try {\n return (await import(spec)) as unknown as EvalRunner;\n } catch (err) {\n throw new Error(\n \"Could not load @databricks/appkit. Run `appkit agent eval` from a \" +\n \"project with @databricks/appkit installed. \" +\n `Cause: ${err instanceof Error ? err.message : String(err)}`,\n );\n }\n}\n\nfunction parseHeaders(values: string[]): Record<string, string> {\n const headers: Record<string, string> = {};\n for (const v of values) {\n const i = v.indexOf(\":\");\n if (i === -1) continue;\n headers[v.slice(0, i).trim()] = v.slice(i + 1).trim();\n }\n return headers;\n}\n\n/**\n * Parse `--min-pass-rate`: a finite number in `[0, 1]`, or `undefined` when\n * unset. Throws on blank, out-of-range, or non-numeric input so a bad gate\n * value fails fast instead of silently disabling or inverting the CI gate.\n */\nexport function parsePassRate(raw: string | undefined): number | undefined {\n if (raw === undefined) return undefined;\n const n = Number(raw);\n // Reject blank too: `Number(\"\")` is 0, which would silently disable the gate.\n if (raw.trim() === \"\" || !Number.isFinite(n) || n < 0 || n > 1) {\n throw new Error(\n `Invalid --min-pass-rate \"${raw}\" — expected a number in [0, 1]`,\n );\n }\n return n;\n}\n\n/** Parse a positive-integer CLI option; junk, zero, or negative → undefined. */\nfunction positiveInt(raw: string | undefined): number | undefined {\n const n = raw ? Number.parseInt(raw, 10) : Number.NaN;\n return n > 0 ? n : undefined;\n}\n\n/** True when `url` answers with any HTTP response (a 404 still proves it's up). */\nasync function isServerUp(url: string): Promise<boolean> {\n try {\n await fetch(url, { signal: AbortSignal.timeout(2000) });\n return true;\n } catch {\n return false;\n }\n}\n\n/**\n * Start the app under test per a root config's `webServer`, à la Playwright:\n * reuse a server already answering at `url` (unless `reuseExisting: false`),\n * else spawn `command`, poll `url` until it answers or `timeoutMs` elapses.\n * Returns a `stop()` that kills the spawned process group (a no-op when the\n * server was reused). Under a machine reporter (`machine`) our logs and the\n * child's stdout are both routed to stderr, so the report stream on stdout\n * stays clean. Throws if the server never comes up.\n */\nasync function startWebServer(\n webServer: NonNullable<EvalConfig[\"webServer\"]>,\n baseUrl: string,\n machine: boolean,\n): Promise<{ stop: () => void }> {\n const url = webServer.url ?? baseUrl;\n const reuse = webServer.reuseExisting !== false;\n const noop = { stop: () => {} };\n\n if (reuse && (await isServerUp(url))) {\n console.error(`Reusing server already running at ${url}`);\n return noop;\n }\n\n console.error(`Starting web server: ${webServer.command}`);\n // `detached` + a negative-PID kill lets us tear down the whole process group\n // (dev servers spawn child processes). Under a machine reporter the child's\n // stdout is sent to our stderr (fd 2) so it can't corrupt the report we later\n // write to stdout; in text mode it inherits so the user sees build output.\n const child: ChildProcess = spawn(webServer.command, {\n shell: true,\n detached: true,\n stdio: machine ? [\"ignore\", 2, \"inherit\"] : \"inherit\",\n });\n\n let exited = false;\n child.on(\"exit\", () => {\n exited = true;\n });\n\n const stop = (): void => {\n if (exited || child.pid === undefined) return;\n try {\n process.kill(-child.pid, \"SIGTERM\");\n } catch {\n // Group already gone, or never became a leader — best-effort.\n }\n };\n\n const deadline = Date.now() + (webServer.timeoutMs ?? 60_000);\n try {\n while (Date.now() < deadline) {\n if (exited) throw new Error(\"web server exited before becoming ready\");\n if (await isServerUp(url)) {\n console.error(`Web server ready at ${url}`);\n return { stop };\n }\n await new Promise((r) => setTimeout(r, 500));\n }\n } catch (err) {\n stop();\n throw err;\n }\n stop();\n throw new Error(\n `web server did not respond at ${url} within ${webServer.timeoutMs ?? 60_000}ms`,\n );\n}\n\ninterface EvalOptions {\n url?: string;\n strict?: boolean;\n root?: string;\n header?: string[];\n tag?: string[];\n profile?: string;\n databricksHost?: string;\n databricksToken?: string;\n experiment?: string;\n judgeModel?: string;\n concurrency?: string;\n warehouseId?: string;\n timeout?: string;\n retries?: string;\n minPassRate?: string;\n reporter?: \"text\" | \"json\" | \"junit\";\n output?: string;\n}\n\n/** Resolved Databricks host + bearer (either field may be absent). */\ntype Auth = { host?: string; token?: string };\n\n/**\n * Native MLflow \"Evaluation run\" config — only when creds + an experiment are\n * all present (traces live in the app; the run + scores are driven from here).\n */\nfunction resolveMlflow(opts: EvalOptions, auth: Auth) {\n const experimentId = opts.experiment ?? process.env.MLFLOW_EXPERIMENT_ID;\n if (!(auth.host && auth.token && experimentId)) return undefined;\n // UC-backed experiments need a SQL warehouse to write assessments to their\n // V4 traces. Mirror mlflow's env var, and accept the common DATABRICKS one.\n const sqlWarehouseId =\n opts.warehouseId ??\n process.env.MLFLOW_TRACING_SQL_WAREHOUSE_ID ??\n process.env.DATABRICKS_WAREHOUSE_ID;\n return {\n host: auth.host,\n token: auth.token,\n experimentId,\n ...(sqlWarehouseId ? { sqlWarehouseId } : {}),\n };\n}\n\n/** LLM-as-judge config — reuses the Databricks creds + a judge serving endpoint. */\nfunction resolveJudge(opts: EvalOptions, auth: Auth) {\n const model = opts.judgeModel ?? process.env.APPKIT_JUDGE_MODEL;\n return model && auth.host && auth.token\n ? { host: auth.host, token: auth.token, model }\n : undefined;\n}\n\n/**\n * Progress reporter: stream each eval as it runs instead of going silent. In a\n * machine reporter (json/junit) the live per-eval streaming is suppressed and\n * banners go to stderr (via `info`), keeping stdout clean for the report.\n */\nfunction makeProgressReporter(\n runner: EvalRunner,\n url: string,\n machine: boolean,\n info: (msg: string) => void,\n): (event: EvalProgress) => void {\n return (event) => {\n switch (event.type) {\n case \"discovered\":\n info(\n `Running ${event.total} eval${event.total === 1 ? \"\" : \"s\"} against ${url}\\n`,\n );\n break;\n case \"run-created\":\n info(`MLflow evaluation run: ${event.runId}\\n`);\n break;\n case \"result\": {\n if (machine) break;\n // One full line per completion — evals run concurrently, so a split\n // \"start … glyph\" prefix would interleave into garbage.\n console.log(\n `[${event.index + 1}/${event.total}] ${runner.formatEvalHeadline(event.result)}`,\n );\n for (const line of runner.formatEvalDetail(event.result)) {\n console.log(line);\n }\n break;\n }\n }\n };\n}\n\nfunction formatFailureLine(f: {\n traceId: string;\n status?: number;\n error?: string;\n}): string {\n return ` ✗ trace ${f.traceId}: ${f.status ?? \"\"} ${f.error ?? \"\"}`.trim();\n}\n\n/**\n * Print the MLflow assessment/finish outcome after a run that created one. The\n * summary line goes through `info` (stderr under a machine reporter); per-trace\n * failures and finish errors always go to stderr.\n */\nfunction printMlflowOutcome(\n mlflow: NonNullable<EvalRunSummary[\"mlflow\"]>,\n info: (msg: string) => void,\n): void {\n const { report, finish } = mlflow;\n info(\n `MLflow: ${report.written} assessment(s) written` +\n (report.skipped ? `, ${report.skipped} skipped` : \"\") +\n (report.failures.length ? `, ${report.failures.length} failed` : \"\"),\n );\n for (const f of report.failures) {\n console.error(formatFailureLine(f));\n }\n if (finish.metricsError) {\n console.error(` ⚠ metrics not logged: ${finish.metricsError}`);\n }\n if (!finish.finished) {\n console.error(\n ` ✗ run left RUNNING — failed to finish: ${finish.finishError ?? \"unknown\"}`,\n );\n }\n}\n\nasync function runAgentEval(\n filter: string | undefined,\n opts: EvalOptions,\n): Promise<void> {\n const runner = await loadRunner();\n\n // Root `evals.config.ts` (project root) carries run-wide settings — baseUrl,\n // webServer, and defaults for concurrency/timeout. A CLI flag always wins.\n const rootDir = opts.root ?? process.cwd();\n const config = (await runner.loadRootEvalConfig(rootDir)) ?? {};\n\n // Base URL: --url flag > config.baseUrl > built-in default. `--url` has no\n // commander default so an unset flag is undefined and lets config win.\n const baseUrl = opts.url ?? config.baseUrl ?? \"http://localhost:8000\";\n\n // Databricks credentials shared by auth resolution and the workspace client:\n // an explicit flag/DATABRICKS_* env wins, else the SDK resolves from the CLI\n // profile.\n const credentials = {\n profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,\n host: opts.databricksHost ?? process.env.DATABRICKS_HOST,\n token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN,\n };\n\n // Resolve Databricks host + bearer the AppKit-native way: an explicit\n // host/token wins; otherwise the SDK mints an OAuth token from the CLI\n // profile — so no hand-set PAT is required.\n const auth: Auth = (await runner.resolveDatabricksAuth(credentials)) ?? {};\n\n // Managed-dataset reads: a workspace client (same profile/host/token) + a SQL\n // warehouse. Only needed by evals that declare `dataset`.\n const warehouseId = opts.warehouseId ?? process.env.DATABRICKS_WAREHOUSE_ID;\n const workspaceClient = runner.resolveWorkspaceClient(credentials);\n\n // Max concurrency: the `--concurrency` flag wins over the root config's value\n // (junk/zero/negative → undefined, so it falls back); else the runner default.\n const concurrency = positiveInt(opts.concurrency) ?? config.maxConcurrency;\n\n // Runner-level default per-eval timeout (ms). --timeout flag wins over the\n // root config; a per-eval `timeoutMs` overrides both (applied in the runner).\n const timeoutMs = positiveInt(opts.timeout) ?? config.timeoutMs;\n\n // Extra attempts for evals that fail on an infra error (turn/timeout). Junk\n // or negative input falls back to no retries.\n const retries = positiveInt(opts.retries);\n\n // Validate up front so a bad gate value fails before the run, not after.\n let minPassRate: number | undefined;\n try {\n minPassRate = parsePassRate(opts.minPassRate);\n } catch (err) {\n console.error(err instanceof Error ? err.message : String(err));\n process.exitCode = 1;\n return;\n }\n\n // In a machine reporter (json/junit), stdout is reserved for the report (it\n // may be piped), so human-facing lines go to stderr and the per-eval live\n // streaming is suppressed. Text mode keeps its current stdout behavior.\n const reporter = opts.reporter ?? \"text\";\n const machine = reporter !== \"text\";\n const info = (msg: string): void => {\n if (machine) console.error(msg);\n else console.log(msg);\n };\n\n // Boot the app under test if the root config declares a webServer (reuses an\n // already-running server unless told otherwise); always torn down after. The\n // boot runs inside the try so a server that never comes up takes the clean\n // error path below instead of escaping as an unhandled rejection.\n let server: { stop: () => void } | undefined;\n let summary: EvalRunSummary;\n try {\n server = config.webServer\n ? await startWebServer(config.webServer, baseUrl, machine)\n : undefined;\n summary = await runner.runEvalsInDir({\n rootDir,\n baseUrl,\n filter,\n tags: opts.tag,\n strict: opts.strict,\n headers: opts.header ? parseHeaders(opts.header) : undefined,\n concurrency,\n mlflow: resolveMlflow(opts, auth),\n judge: resolveJudge(opts, auth),\n workspaceClient,\n warehouseId,\n timeoutMs,\n retries,\n onEvent: makeProgressReporter(runner, baseUrl, machine, info),\n });\n } catch (err) {\n // Setup failures — a web server that never comes up, or e.g. a bad\n // --experiment for the MLflow run — reject before any eval runs; surface a\n // clean message + non-zero exit rather than an unhandled promise rejection\n // with a raw stack.\n console.error(\n `\\nEval run failed: ${err instanceof Error ? err.message : String(err)}`,\n );\n process.exitCode = 1;\n return;\n } finally {\n server?.stop();\n }\n\n // The final human summary always shows (stderr for machine reporters so it\n // never pollutes the report on stdout/file).\n info(`\\n${runner.formatSummaryLine(summary.results)}`);\n\n if (summary.mlflow) {\n printMlflowOutcome(summary.mlflow, info);\n } else {\n info(\n \"\\nMLflow evaluation run skipped — pass --experiment (or set\" +\n \" MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.\",\n );\n }\n\n // Machine-readable report: build the string with a pure formatter, then emit\n // it to --output <file> or stdout (kept clean of the human noise above).\n if (machine) {\n const report =\n reporter === \"json\"\n ? runner.formatResultsJson(summary.results)\n : runner.formatResultsJUnit(summary.results);\n if (opts.output) {\n try {\n fs.writeFileSync(opts.output, `${report}\\n`);\n } catch (err) {\n console.error(\n `Failed to write ${reporter} report to ${opts.output}: ${\n err instanceof Error ? err.message : String(err)\n }`,\n );\n process.exitCode = 1;\n return;\n }\n info(`Wrote ${reporter} report to ${opts.output}`);\n } else {\n process.stdout.write(`${report}\\n`);\n }\n }\n\n const stats = runner.summarize(summary.results);\n if (minPassRate !== undefined) {\n // Threshold mode: gate on the aggregate pass rate rather than requiring\n // every eval to pass.\n const ok = stats.passRate >= minPassRate;\n info(\n `Pass rate ${(stats.passRate * 100).toFixed(0)}% (threshold ${(\n minPassRate * 100\n ).toFixed(0)}%) — ${ok ? \"OK\" : \"below threshold\"}`,\n );\n if (!ok) process.exitCode = 1;\n } else if (!stats.allPassed) {\n process.exitCode = 1;\n }\n}\n\nexport const agentEvalCommand = new Command(\"eval\")\n .description(\n \"Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app\",\n )\n .argument(\n \"[filter]\",\n \"Only run evals whose <agent>/<id> contains this substring (or an exact agent id)\",\n )\n .option(\n \"--url <url>\",\n \"Base URL of the app to drive (default: evals.config.ts baseUrl, else http://localhost:8000)\",\n )\n .option(\"--strict\", \"Fail on soft-assertion misses too\", false)\n .option(\n \"--concurrency <n>\",\n \"Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)\",\n )\n .option(\n \"--root <dir>\",\n \"Project root containing server/agents/ (default: cwd)\",\n )\n .option(\n \"--header <header...>\",\n \"Extra request header as 'Key: value' (repeatable)\",\n )\n .option(\n \"--tag <tag...>\",\n \"Only run evals tagged with one of these tags (repeatable)\",\n )\n .option(\n \"--profile <name>\",\n \"Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)\",\n )\n .option(\n \"--databricks-host <host>\",\n \"Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)\",\n )\n .option(\n \"--databricks-token <token>\",\n \"Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)\",\n )\n .option(\n \"--experiment <id>\",\n \"MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)\",\n )\n .option(\n \"--warehouse-id <id>\",\n \"SQL warehouse id for reading managed eval datasets and writing assessments to UC-backed experiments (default: DATABRICKS_WAREHOUSE_ID, or MLFLOW_TRACING_SQL_WAREHOUSE_ID for assessments)\",\n )\n .option(\n \"--judge-model <endpoint>\",\n \"Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)\",\n )\n .option(\n \"--timeout <ms>\",\n \"Default per-eval timeout in ms (a per-eval timeoutMs overrides it)\",\n )\n .option(\n \"--retries <n>\",\n \"Re-run an eval up to N times when it fails on an infra error (turn/timeout); assertion failures are not retried\",\n )\n .option(\n \"--min-pass-rate <rate>\",\n \"Gate on aggregate pass rate (0..1) instead of requiring every eval to pass; exit 1 when below\",\n )\n .addOption(\n new Option(\n \"--reporter <format>\",\n \"Report format: text (live console), json (dashboards), or junit (CI test reporters)\",\n )\n .choices([\"text\", \"json\", \"junit\"])\n .default(\"text\"),\n )\n .option(\n \"--output <file>\",\n \"Write the json/junit report to this file instead of stdout (ignored for text)\",\n )\n .action(runAgentEval);\n"],"mappings":";;;;;;;;;;AAqFA,eAAe,aAAkC;CAC/C,MAAM,OAAO;AACb,KAAI;AACF,SAAQ,MAAM,OAAO;UACd,KAAK;AACZ,QAAM,IAAI,MACR,yHAEY,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAC7D;;;AAIL,SAAS,aAAa,QAA0C;CAC9D,MAAM,UAAkC,EAAE;AAC1C,MAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,IAAI,EAAE,QAAQ,IAAI;AACxB,MAAI,MAAM,GAAI;AACd,UAAQ,EAAE,MAAM,GAAG,EAAE,CAAC,MAAM,IAAI,EAAE,MAAM,IAAI,EAAE,CAAC,MAAM;;AAEvD,QAAO;;;;;;;AAQT,SAAgB,cAAc,KAA6C;AACzE,KAAI,QAAQ,OAAW,QAAO;CAC9B,MAAM,IAAI,OAAO,IAAI;AAErB,KAAI,IAAI,MAAM,KAAK,MAAM,CAAC,OAAO,SAAS,EAAE,IAAI,IAAI,KAAK,IAAI,EAC3D,OAAM,IAAI,MACR,4BAA4B,IAAI,iCACjC;AAEH,QAAO;;;AAIT,SAAS,YAAY,KAA6C;CAChE,MAAM,IAAI,MAAM,OAAO,SAAS,KAAK,GAAG,GAAG;AAC3C,QAAO,IAAI,IAAI,IAAI;;;AAIrB,eAAe,WAAW,KAA+B;AACvD,KAAI;AACF,QAAM,MAAM,KAAK,EAAE,QAAQ,YAAY,QAAQ,IAAK,EAAE,CAAC;AACvD,SAAO;SACD;AACN,SAAO;;;;;;;;;;;;AAaX,eAAe,eACb,WACA,SACA,SAC+B;CAC/B,MAAM,MAAM,UAAU,OAAO;CAC7B,MAAM,QAAQ,UAAU,kBAAkB;CAC1C,MAAM,OAAO,EAAE,YAAY,IAAI;AAE/B,KAAI,SAAU,MAAM,WAAW,IAAI,EAAG;AACpC,UAAQ,MAAM,qCAAqC,MAAM;AACzD,SAAO;;AAGT,SAAQ,MAAM,wBAAwB,UAAU,UAAU;CAK1D,MAAM,QAAsB,MAAM,UAAU,SAAS;EACnD,OAAO;EACP,UAAU;EACV,OAAO,UAAU;GAAC;GAAU;GAAG;GAAU,GAAG;EAC7C,CAAC;CAEF,IAAI,SAAS;AACb,OAAM,GAAG,cAAc;AACrB,WAAS;GACT;CAEF,MAAM,aAAmB;AACvB,MAAI,UAAU,MAAM,QAAQ,OAAW;AACvC,MAAI;AACF,WAAQ,KAAK,CAAC,MAAM,KAAK,UAAU;UAC7B;;CAKV,MAAM,WAAW,KAAK,KAAK,IAAI,UAAU,aAAa;AACtD,KAAI;AACF,SAAO,KAAK,KAAK,GAAG,UAAU;AAC5B,OAAI,OAAQ,OAAM,IAAI,MAAM,0CAA0C;AACtE,OAAI,MAAM,WAAW,IAAI,EAAE;AACzB,YAAQ,MAAM,uBAAuB,MAAM;AAC3C,WAAO,EAAE,MAAM;;AAEjB,SAAM,IAAI,SAAS,MAAM,WAAW,GAAG,IAAI,CAAC;;UAEvC,KAAK;AACZ,QAAM;AACN,QAAM;;AAER,OAAM;AACN,OAAM,IAAI,MACR,iCAAiC,IAAI,UAAU,UAAU,aAAa,IAAO,IAC9E;;;;;;AA8BH,SAAS,cAAc,MAAmB,MAAY;CACpD,MAAM,eAAe,KAAK,cAAc,QAAQ,IAAI;AACpD,KAAI,EAAE,KAAK,QAAQ,KAAK,SAAS,cAAe,QAAO;CAGvD,MAAM,iBACJ,KAAK,eACL,QAAQ,IAAI,mCACZ,QAAQ,IAAI;AACd,QAAO;EACL,MAAM,KAAK;EACX,OAAO,KAAK;EACZ;EACA,GAAI,iBAAiB,EAAE,gBAAgB,GAAG,EAAE;EAC7C;;;AAIH,SAAS,aAAa,MAAmB,MAAY;CACnD,MAAM,QAAQ,KAAK,cAAc,QAAQ,IAAI;AAC7C,QAAO,SAAS,KAAK,QAAQ,KAAK,QAC9B;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;EAAO;EAAO,GAC7C;;;;;;;AAQN,SAAS,qBACP,QACA,KACA,SACA,MAC+B;AAC/B,SAAQ,UAAU;AAChB,UAAQ,MAAM,MAAd;GACE,KAAK;AACH,SACE,WAAW,MAAM,MAAM,OAAO,MAAM,UAAU,IAAI,KAAK,IAAI,WAAW,IAAI,IAC3E;AACD;GACF,KAAK;AACH,SAAK,0BAA0B,MAAM,MAAM,IAAI;AAC/C;GACF,KAAK;AACH,QAAI,QAAS;AAGb,YAAQ,IACN,IAAI,MAAM,QAAQ,EAAE,GAAG,MAAM,MAAM,IAAI,OAAO,mBAAmB,MAAM,OAAO,GAC/E;AACD,SAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,OAAO,CACtD,SAAQ,IAAI,KAAK;AAEnB;;;;AAMR,SAAS,kBAAkB,GAIhB;AACT,QAAO,aAAa,EAAE,QAAQ,IAAI,EAAE,UAAU,GAAG,GAAG,EAAE,SAAS,KAAK,MAAM;;;;;;;AAQ5E,SAAS,mBACP,QACA,MACM;CACN,MAAM,EAAE,QAAQ,WAAW;AAC3B,MACE,WAAW,OAAO,QAAQ,2BACvB,OAAO,UAAU,KAAK,OAAO,QAAQ,YAAY,OACjD,OAAO,SAAS,SAAS,KAAK,OAAO,SAAS,OAAO,WAAW,IACpE;AACD,MAAK,MAAM,KAAK,OAAO,SACrB,SAAQ,MAAM,kBAAkB,EAAE,CAAC;AAErC,KAAI,OAAO,aACT,SAAQ,MAAM,2BAA2B,OAAO,eAAe;AAEjE,KAAI,CAAC,OAAO,SACV,SAAQ,MACN,4CAA4C,OAAO,eAAe,YACnE;;AAIL,eAAe,aACb,QACA,MACe;CACf,MAAM,SAAS,MAAM,YAAY;CAIjC,MAAM,UAAU,KAAK,QAAQ,QAAQ,KAAK;CAC1C,MAAM,SAAU,MAAM,OAAO,mBAAmB,QAAQ,IAAK,EAAE;CAI/D,MAAM,UAAU,KAAK,OAAO,OAAO,WAAW;CAK9C,MAAM,cAAc;EAClB,SAAS,KAAK,WAAW,QAAQ,IAAI;EACrC,MAAM,KAAK,kBAAkB,QAAQ,IAAI;EACzC,OAAO,KAAK,mBAAmB,QAAQ,IAAI;EAC5C;CAKD,MAAM,OAAc,MAAM,OAAO,sBAAsB,YAAY,IAAK,EAAE;CAI1E,MAAM,cAAc,KAAK,eAAe,QAAQ,IAAI;CACpD,MAAM,kBAAkB,OAAO,uBAAuB,YAAY;CAIlE,MAAM,cAAc,YAAY,KAAK,YAAY,IAAI,OAAO;CAI5D,MAAM,YAAY,YAAY,KAAK,QAAQ,IAAI,OAAO;CAItD,MAAM,UAAU,YAAY,KAAK,QAAQ;CAGzC,IAAI;AACJ,KAAI;AACF,gBAAc,cAAc,KAAK,YAAY;UACtC,KAAK;AACZ,UAAQ,MAAM,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,CAAC;AAC/D,UAAQ,WAAW;AACnB;;CAMF,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,UAAU,aAAa;CAC7B,MAAM,QAAQ,QAAsB;AAClC,MAAI,QAAS,SAAQ,MAAM,IAAI;MAC1B,SAAQ,IAAI,IAAI;;CAOvB,IAAI;CACJ,IAAI;AACJ,KAAI;AACF,WAAS,OAAO,YACZ,MAAM,eAAe,OAAO,WAAW,SAAS,QAAQ,GACxD;AACJ,YAAU,MAAM,OAAO,cAAc;GACnC;GACA;GACA;GACA,MAAM,KAAK;GACX,QAAQ,KAAK;GACb,SAAS,KAAK,SAAS,aAAa,KAAK,OAAO,GAAG;GACnD;GACA,QAAQ,cAAc,MAAM,KAAK;GACjC,OAAO,aAAa,MAAM,KAAK;GAC/B;GACA;GACA;GACA;GACA,SAAS,qBAAqB,QAAQ,SAAS,SAAS,KAAK;GAC9D,CAAC;UACK,KAAK;AAKZ,UAAQ,MACN,sBAAsB,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GACvE;AACD,UAAQ,WAAW;AACnB;WACQ;AACR,UAAQ,MAAM;;AAKhB,MAAK,KAAK,OAAO,kBAAkB,QAAQ,QAAQ,GAAG;AAEtD,KAAI,QAAQ,OACV,oBAAmB,QAAQ,QAAQ,KAAK;KAExC,MACE,oIAED;AAKH,KAAI,SAAS;EACX,MAAM,SACJ,aAAa,SACT,OAAO,kBAAkB,QAAQ,QAAQ,GACzC,OAAO,mBAAmB,QAAQ,QAAQ;AAChD,MAAI,KAAK,QAAQ;AACf,OAAI;AACF,OAAG,cAAc,KAAK,QAAQ,GAAG,OAAO,IAAI;YACrC,KAAK;AACZ,YAAQ,MACN,mBAAmB,SAAS,aAAa,KAAK,OAAO,IACnD,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAEnD;AACD,YAAQ,WAAW;AACnB;;AAEF,QAAK,SAAS,SAAS,aAAa,KAAK,SAAS;QAElD,SAAQ,OAAO,MAAM,GAAG,OAAO,IAAI;;CAIvC,MAAM,QAAQ,OAAO,UAAU,QAAQ,QAAQ;AAC/C,KAAI,gBAAgB,QAAW;EAG7B,MAAM,KAAK,MAAM,YAAY;AAC7B,OACE,cAAc,MAAM,WAAW,KAAK,QAAQ,EAAE,CAAC,gBAC7C,cAAc,KACd,QAAQ,EAAE,CAAC,OAAO,KAAK,OAAO,oBACjC;AACD,MAAI,CAAC,GAAI,SAAQ,WAAW;YACnB,CAAC,MAAM,UAChB,SAAQ,WAAW;;AAIvB,MAAa,mBAAmB,IAAI,QAAQ,OAAO,CAChD,YACC,6EACD,CACA,SACC,YACA,mFACD,CACA,OACC,eACA,8FACD,CACA,OAAO,YAAY,qCAAqC,MAAM,CAC9D,OACC,qBACA,wGACD,CACA,OACC,gBACA,wDACD,CACA,OACC,wBACA,oDACD,CACA,OACC,kBACA,4DACD,CACA,OACC,oBACA,6FACD,CACA,OACC,4BACA,4EACD,CACA,OACC,8BACA,8EACD,CACA,OACC,qBACA,8EACD,CACA,OACC,uBACA,6LACD,CACA,OACC,4BACA,kGACD,CACA,OACC,kBACA,qEACD,CACA,OACC,iBACA,kHACD,CACA,OACC,0BACA,gGACD,CACA,UACC,IAAI,OACF,uBACA,sFACD,CACE,QAAQ;CAAC;CAAQ;CAAQ;CAAQ,CAAC,CAClC,QAAQ,OAAO,CACnB,CACA,OACC,mBACA,gFACD,CACA,OAAO,aAAa"}
|
|
@@ -7,9 +7,9 @@ import { parseEnv } from "./env-reconcile.js";
|
|
|
7
7
|
import fs from "node:fs";
|
|
8
8
|
import path from "node:path";
|
|
9
9
|
import { Command } from "commander";
|
|
10
|
+
import { spawnSync } from "node:child_process";
|
|
10
11
|
import process from "node:process";
|
|
11
12
|
import pc from "picocolors";
|
|
12
|
-
import { spawnSync } from "node:child_process";
|
|
13
13
|
|
|
14
14
|
//#region src/cli/commands/registry/add.ts
|
|
15
15
|
/** Subdirectories that commonly hold the frontend / server in an AppKit app. */
|
|
@@ -2,8 +2,8 @@ import { APP_YAML_FILE, DATABRICKS_YML_FILE, bindingToNode } from "../../deploy-
|
|
|
2
2
|
import { planHasContent } from "./config-plan.js";
|
|
3
3
|
import fs from "node:fs";
|
|
4
4
|
import path from "node:path";
|
|
5
|
-
import pc from "picocolors";
|
|
6
5
|
import { spawnSync } from "node:child_process";
|
|
6
|
+
import pc from "picocolors";
|
|
7
7
|
import { parseDocument } from "yaml";
|
|
8
8
|
|
|
9
9
|
//#region src/cli/commands/registry/config-writer.ts
|
package/dist/cli/index.js
CHANGED
|
@@ -30,7 +30,10 @@ cmd.addCommand(doctorCommand);
|
|
|
30
30
|
cmd.addCommand(registryCommand, { hidden: true });
|
|
31
31
|
cmd.addCommand(addCommand, { hidden: true });
|
|
32
32
|
cmd.addCommand(agentCommand);
|
|
33
|
-
await cmd.parseAsync()
|
|
33
|
+
await cmd.parseAsync().catch((err) => {
|
|
34
|
+
console.error(`\n${err instanceof Error ? err.message : String(err)}`);
|
|
35
|
+
process.exitCode = 1;
|
|
36
|
+
});
|
|
34
37
|
|
|
35
38
|
//#endregion
|
|
36
39
|
export { };
|
package/dist/cli/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/cli/index.ts"],"sourcesContent":["#!/usr/bin/env node\nimport \"dotenv/config\";\nimport { readFileSync } from \"node:fs\";\nimport { dirname, join } from \"node:path\";\nimport { fileURLToPath } from \"node:url\";\n\nimport { Command } from \"commander\";\n\nimport { agentCommand } from \"./commands/agent/index.js\";\nimport { codemodCommand } from \"./commands/codemod/index.js\";\nimport { docsCommand } from \"./commands/docs.js\";\nimport { doctorCommand } from \"./commands/doctor/index.js\";\nimport { generateTypesCommand } from \"./commands/generate-types.js\";\nimport { lintCommand } from \"./commands/lint.js\";\nimport { pluginCommand } from \"./commands/plugin/index.js\";\nimport { addCommand } from \"./commands/registry/add.js\";\nimport { registryCommand } from \"./commands/registry/index.js\";\nimport { setupCommand } from \"./commands/setup.js\";\n\nconst __dirname = dirname(fileURLToPath(import.meta.url));\nconst pkgPath = join(__dirname, \"../../package.json\");\nconst pkg = JSON.parse(readFileSync(pkgPath, \"utf-8\"));\n\nconst cmd = new Command();\n\ncmd\n .name(\"appkit\")\n .description(\"CLI tools for Databricks AppKit\")\n .version(pkg.version);\n\ncmd.addCommand(setupCommand);\ncmd.addCommand(generateTypesCommand);\ncmd.addCommand(lintCommand);\ncmd.addCommand(docsCommand);\ncmd.addCommand(pluginCommand);\ncmd.addCommand(codemodCommand);\ncmd.addCommand(doctorCommand);\n// Registry commands are executable but hidden from --help while the feature\n// is still in development (registry + add work end-to-end but aren't announced).\ncmd.addCommand(registryCommand, { hidden: true });\ncmd.addCommand(addCommand, { hidden: true });\ncmd.addCommand(agentCommand);\n\nawait cmd.parseAsync();\n"],"mappings":";;;;;;;;;;;;;;;;;;AAoBA,MAAM,UAAU,KADE,QAAQ,cAAc,OAAO,KAAK,IAAI,CAAC,EACzB,qBAAqB;AACrD,MAAM,MAAM,KAAK,MAAM,aAAa,SAAS,QAAQ,CAAC;AAEtD,MAAM,MAAM,IAAI,SAAS;AAEzB,IACG,KAAK,SAAS,CACd,YAAY,kCAAkC,CAC9C,QAAQ,IAAI,QAAQ;AAEvB,IAAI,WAAW,aAAa;AAC5B,IAAI,WAAW,qBAAqB;AACpC,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,cAAc;AAC7B,IAAI,WAAW,eAAe;AAC9B,IAAI,WAAW,cAAc;AAG7B,IAAI,WAAW,iBAAiB,EAAE,QAAQ,MAAM,CAAC;AACjD,IAAI,WAAW,YAAY,EAAE,QAAQ,MAAM,CAAC;AAC5C,IAAI,WAAW,aAAa;
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/cli/index.ts"],"sourcesContent":["#!/usr/bin/env node\nimport \"dotenv/config\";\nimport { readFileSync } from \"node:fs\";\nimport { dirname, join } from \"node:path\";\nimport { fileURLToPath } from \"node:url\";\n\nimport { Command } from \"commander\";\n\nimport { agentCommand } from \"./commands/agent/index.js\";\nimport { codemodCommand } from \"./commands/codemod/index.js\";\nimport { docsCommand } from \"./commands/docs.js\";\nimport { doctorCommand } from \"./commands/doctor/index.js\";\nimport { generateTypesCommand } from \"./commands/generate-types.js\";\nimport { lintCommand } from \"./commands/lint.js\";\nimport { pluginCommand } from \"./commands/plugin/index.js\";\nimport { addCommand } from \"./commands/registry/add.js\";\nimport { registryCommand } from \"./commands/registry/index.js\";\nimport { setupCommand } from \"./commands/setup.js\";\n\nconst __dirname = dirname(fileURLToPath(import.meta.url));\nconst pkgPath = join(__dirname, \"../../package.json\");\nconst pkg = JSON.parse(readFileSync(pkgPath, \"utf-8\"));\n\nconst cmd = new Command();\n\ncmd\n .name(\"appkit\")\n .description(\"CLI tools for Databricks AppKit\")\n .version(pkg.version);\n\ncmd.addCommand(setupCommand);\ncmd.addCommand(generateTypesCommand);\ncmd.addCommand(lintCommand);\ncmd.addCommand(docsCommand);\ncmd.addCommand(pluginCommand);\ncmd.addCommand(codemodCommand);\ncmd.addCommand(doctorCommand);\n// Registry commands are executable but hidden from --help while the feature\n// is still in development (registry + add work end-to-end but aren't announced).\ncmd.addCommand(registryCommand, { hidden: true });\ncmd.addCommand(addCommand, { hidden: true });\ncmd.addCommand(agentCommand);\n\n// Surface a clean message for any uncaught setup failure (e.g. a root\n// evals.config.ts that throws on import) instead of a raw unhandled-rejection\n// stack; a non-zero exit still gates CI.\nawait cmd.parseAsync().catch((err: unknown) => {\n console.error(`\\n${err instanceof Error ? err.message : String(err)}`);\n process.exitCode = 1;\n});\n"],"mappings":";;;;;;;;;;;;;;;;;;AAoBA,MAAM,UAAU,KADE,QAAQ,cAAc,OAAO,KAAK,IAAI,CAAC,EACzB,qBAAqB;AACrD,MAAM,MAAM,KAAK,MAAM,aAAa,SAAS,QAAQ,CAAC;AAEtD,MAAM,MAAM,IAAI,SAAS;AAEzB,IACG,KAAK,SAAS,CACd,YAAY,kCAAkC,CAC9C,QAAQ,IAAI,QAAQ;AAEvB,IAAI,WAAW,aAAa;AAC5B,IAAI,WAAW,qBAAqB;AACpC,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,cAAc;AAC7B,IAAI,WAAW,eAAe;AAC9B,IAAI,WAAW,cAAc;AAG7B,IAAI,WAAW,iBAAiB,EAAE,QAAQ,MAAM,CAAC;AACjD,IAAI,WAAW,YAAY,EAAE,QAAQ,MAAM,CAAC;AAC5C,IAAI,WAAW,aAAa;AAK5B,MAAM,IAAI,YAAY,CAAC,OAAO,QAAiB;AAC7C,SAAQ,MAAM,KAAK,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAAG;AACtE,SAAQ,WAAW;EACnB"}
|
package/dist/evals/discover.d.ts
CHANGED
|
@@ -22,6 +22,13 @@ interface DiscoveredEvalConfig {
|
|
|
22
22
|
* path relative to that evals dir with `.eval.ts` stripped. Sorted + stable.
|
|
23
23
|
*/
|
|
24
24
|
declare function discoverEvalFiles(rootDir: string): DiscoveredEval[];
|
|
25
|
+
/**
|
|
26
|
+
* Path to the root `evals.config.ts` (from {@link defineEvalConfig}) at
|
|
27
|
+
* `<rootDir>/evals.config.ts`, or `undefined` when absent. The root config
|
|
28
|
+
* holds run-wide settings (`baseUrl`, `webServer`); it's distinct from the
|
|
29
|
+
* per-agent configs found by {@link discoverEvalConfigs}.
|
|
30
|
+
*/
|
|
31
|
+
declare function findRootEvalConfig(rootDir: string): string | undefined;
|
|
25
32
|
/**
|
|
26
33
|
* Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at
|
|
27
34
|
* `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:
|
|
@@ -30,5 +37,5 @@ declare function discoverEvalFiles(rootDir: string): DiscoveredEval[];
|
|
|
30
37
|
*/
|
|
31
38
|
declare function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[];
|
|
32
39
|
//#endregion
|
|
33
|
-
export { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles };
|
|
40
|
+
export { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig };
|
|
34
41
|
//# sourceMappingURL=discover.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"discover.d.ts","names":[],"sources":["../../src/evals/discover.ts"],"mappings":";;UAOiB,cAAA;EAAc;EAE7B,IAAA;EAF6B;EAI7B,EAAA;EAAA;EAEA,KAAA;AAAA;;UAIe,oBAAA;EAAoB;EAEnC,IAAA;EAAA;EAEA,KAAA;AAAA;;;;;AA+DF;;iBA3BgB,iBAAA,CAAkB,OAAA,WAAkB,cAAA;;;;;;;
|
|
1
|
+
{"version":3,"file":"discover.d.ts","names":[],"sources":["../../src/evals/discover.ts"],"mappings":";;UAOiB,cAAA;EAAc;EAE7B,IAAA;EAF6B;EAI7B,EAAA;EAAA;EAEA,KAAA;AAAA;;UAIe,oBAAA;EAAoB;EAEnC,IAAA;EAAA;EAEA,KAAA;AAAA;;;;;AA+DF;;iBA3BgB,iBAAA,CAAkB,OAAA,WAAkB,cAAA;;;AAsCpD;;;;iBAXgB,kBAAA,CAAmB,OAAA;;;;;;;iBAWnB,mBAAA,CAAoB,OAAA,WAAkB,oBAAA"}
|
package/dist/evals/discover.js
CHANGED
|
@@ -56,6 +56,16 @@ function discoverEvalFiles(rootDir) {
|
|
|
56
56
|
return out.sort((a, b) => a.agent.localeCompare(b.agent) || a.id.localeCompare(b.id));
|
|
57
57
|
}
|
|
58
58
|
/**
|
|
59
|
+
* Path to the root `evals.config.ts` (from {@link defineEvalConfig}) at
|
|
60
|
+
* `<rootDir>/evals.config.ts`, or `undefined` when absent. The root config
|
|
61
|
+
* holds run-wide settings (`baseUrl`, `webServer`); it's distinct from the
|
|
62
|
+
* per-agent configs found by {@link discoverEvalConfigs}.
|
|
63
|
+
*/
|
|
64
|
+
function findRootEvalConfig(rootDir) {
|
|
65
|
+
const file = path.join(rootDir, "evals.config.ts");
|
|
66
|
+
return existsSync(file) ? file : void 0;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
59
69
|
* Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at
|
|
60
70
|
* `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:
|
|
61
71
|
* each agent's config applies only to that agent's evals. Agents without a
|
|
@@ -75,5 +85,5 @@ function discoverEvalConfigs(rootDir) {
|
|
|
75
85
|
}
|
|
76
86
|
|
|
77
87
|
//#endregion
|
|
78
|
-
export { discoverEvalConfigs, discoverEvalFiles };
|
|
88
|
+
export { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig };
|
|
79
89
|
//# sourceMappingURL=discover.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"discover.js","names":[],"sources":["../../src/evals/discover.ts"],"sourcesContent":["import { existsSync, readdirSync } from \"node:fs\";\nimport path from \"node:path\";\n\nimport { agentDirNames } from \"../core/agent/agent-dirs\";\nimport { CODE_AGENTS_SOURCE_DIR } from \"../core/agent/load-code-agents\";\n\n/** An eval file found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEval {\n /** Absolute path to the `*.eval.ts` file. */\n file: string;\n /** Id relative to the agent's evals dir, without `.eval.ts` (e.g. `weather/basic`). */\n id: string;\n /** The agent id (the `server/agents/<agent>` directory name). */\n agent: string;\n}\n\n/** A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEvalConfig {\n /** Absolute path to the `evals.config.ts` file. */\n file: string;\n /** The agent id whose evals this config applies to. */\n agent: string;\n}\n\n/** Recursively collect `*.eval.ts` files under `dir`. Empty when `dir` is absent. */\nfunction evalFilesIn(dir: string): string[] {\n try {\n return readdirSync(dir, { recursive: true, withFileTypes: true })\n .filter((e) => e.isFile() && e.name.endsWith(\".eval.ts\"))\n .map((e) => path.join(e.parentPath, e.name));\n } catch {\n return [];\n }\n}\n\n/**\n * List the agent directory names under `<rootDir>/server/agents/` (empty if the\n * dir is absent). Shared by the eval-file and eval-config discovery below.\n */\nfunction listAgents(rootDir: string): { agentsDir: string; agents: string[] } {\n const agentsDir = path.join(rootDir, CODE_AGENTS_SOURCE_DIR);\n try {\n return {\n agentsDir,\n agents: agentDirNames(readdirSync(agentsDir, { withFileTypes: true })),\n };\n } catch {\n return { agentsDir, agents: [] };\n }\n}\n\n/**\n * Discover evals under `<rootDir>/server/agents/<agent>/evals/` — co-located\n * with each agent's `agent.{md,ts}` (same folder-per-agent layout the agents\n * plugin discovers). The agent id is the folder name; the eval id is the file\n * path relative to that evals dir with `.eval.ts` stripped. Sorted + stable.\n */\nexport function discoverEvalFiles(rootDir: string): DiscoveredEval[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEval[] = [];\n\n for (const agent of agents) {\n const evalsDir = path.join(agentsDir, agent, \"evals\");\n for (const file of evalFilesIn(evalsDir)) {\n const id = path\n .relative(evalsDir, file)\n .replace(/\\.eval\\.ts$/, \"\")\n .split(path.sep)\n .join(\"/\");\n out.push({ file, id, agent });\n }\n }\n\n return out.sort(\n (a, b) => a.agent.localeCompare(b.agent) || a.id.localeCompare(b.id),\n );\n}\n\n/**\n * Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at\n * `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:\n * each agent's config applies only to that agent's evals. Agents without a\n * config file are omitted. Returns a stable, sorted list.\n */\nexport function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEvalConfig[] = [];\n\n for (const agent of agents) {\n const file = path.join(agentsDir, agent, \"evals\", \"evals.config.ts\");\n if (existsSync(file)) out.push({ file, agent });\n }\n\n return out.sort((a, b) => a.agent.localeCompare(b.agent));\n}\n"],"mappings":";;;;;;;AAyBA,SAAS,YAAY,KAAuB;AAC1C,KAAI;AACF,SAAO,YAAY,KAAK;GAAE,WAAW;GAAM,eAAe;GAAM,CAAC,CAC9D,QAAQ,MAAM,EAAE,QAAQ,IAAI,EAAE,KAAK,SAAS,WAAW,CAAC,CACxD,KAAK,MAAM,KAAK,KAAK,EAAE,YAAY,EAAE,KAAK,CAAC;SACxC;AACN,SAAO,EAAE;;;;;;;AAQb,SAAS,WAAW,SAA0D;CAC5E,MAAM,YAAY,KAAK,KAAK,SAAS,uBAAuB;AAC5D,KAAI;AACF,SAAO;GACL;GACA,QAAQ,cAAc,YAAY,WAAW,EAAE,eAAe,MAAM,CAAC,CAAC;GACvE;SACK;AACN,SAAO;GAAE;GAAW,QAAQ,EAAE;GAAE;;;;;;;;;AAUpC,SAAgB,kBAAkB,SAAmC;CACnE,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAAwB,EAAE;AAEhC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,WAAW,KAAK,KAAK,WAAW,OAAO,QAAQ;AACrD,OAAK,MAAM,QAAQ,YAAY,SAAS,EAAE;GACxC,MAAM,KAAK,KACR,SAAS,UAAU,KAAK,CACxB,QAAQ,eAAe,GAAG,CAC1B,MAAM,KAAK,IAAI,CACf,KAAK,IAAI;AACZ,OAAI,KAAK;IAAE;IAAM;IAAI;IAAO,CAAC;;;AAIjC,QAAO,IAAI,MACR,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,IAAI,EAAE,GAAG,cAAc,EAAE,GAAG,CACrE;;;;;;;;AASH,SAAgB,oBAAoB,SAAyC;CAC3E,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAA8B,EAAE;AAEtC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,OAAO,KAAK,KAAK,WAAW,OAAO,SAAS,kBAAkB;AACpE,MAAI,WAAW,KAAK,CAAE,KAAI,KAAK;GAAE;GAAM;GAAO,CAAC;;AAGjD,QAAO,IAAI,MAAM,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,CAAC"}
|
|
1
|
+
{"version":3,"file":"discover.js","names":[],"sources":["../../src/evals/discover.ts"],"sourcesContent":["import { existsSync, readdirSync } from \"node:fs\";\nimport path from \"node:path\";\n\nimport { agentDirNames } from \"../core/agent/agent-dirs\";\nimport { CODE_AGENTS_SOURCE_DIR } from \"../core/agent/load-code-agents\";\n\n/** An eval file found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEval {\n /** Absolute path to the `*.eval.ts` file. */\n file: string;\n /** Id relative to the agent's evals dir, without `.eval.ts` (e.g. `weather/basic`). */\n id: string;\n /** The agent id (the `server/agents/<agent>` directory name). */\n agent: string;\n}\n\n/** A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEvalConfig {\n /** Absolute path to the `evals.config.ts` file. */\n file: string;\n /** The agent id whose evals this config applies to. */\n agent: string;\n}\n\n/** Recursively collect `*.eval.ts` files under `dir`. Empty when `dir` is absent. */\nfunction evalFilesIn(dir: string): string[] {\n try {\n return readdirSync(dir, { recursive: true, withFileTypes: true })\n .filter((e) => e.isFile() && e.name.endsWith(\".eval.ts\"))\n .map((e) => path.join(e.parentPath, e.name));\n } catch {\n return [];\n }\n}\n\n/**\n * List the agent directory names under `<rootDir>/server/agents/` (empty if the\n * dir is absent). Shared by the eval-file and eval-config discovery below.\n */\nfunction listAgents(rootDir: string): { agentsDir: string; agents: string[] } {\n const agentsDir = path.join(rootDir, CODE_AGENTS_SOURCE_DIR);\n try {\n return {\n agentsDir,\n agents: agentDirNames(readdirSync(agentsDir, { withFileTypes: true })),\n };\n } catch {\n return { agentsDir, agents: [] };\n }\n}\n\n/**\n * Discover evals under `<rootDir>/server/agents/<agent>/evals/` — co-located\n * with each agent's `agent.{md,ts}` (same folder-per-agent layout the agents\n * plugin discovers). The agent id is the folder name; the eval id is the file\n * path relative to that evals dir with `.eval.ts` stripped. Sorted + stable.\n */\nexport function discoverEvalFiles(rootDir: string): DiscoveredEval[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEval[] = [];\n\n for (const agent of agents) {\n const evalsDir = path.join(agentsDir, agent, \"evals\");\n for (const file of evalFilesIn(evalsDir)) {\n const id = path\n .relative(evalsDir, file)\n .replace(/\\.eval\\.ts$/, \"\")\n .split(path.sep)\n .join(\"/\");\n out.push({ file, id, agent });\n }\n }\n\n return out.sort(\n (a, b) => a.agent.localeCompare(b.agent) || a.id.localeCompare(b.id),\n );\n}\n\n/**\n * Path to the root `evals.config.ts` (from {@link defineEvalConfig}) at\n * `<rootDir>/evals.config.ts`, or `undefined` when absent. The root config\n * holds run-wide settings (`baseUrl`, `webServer`); it's distinct from the\n * per-agent configs found by {@link discoverEvalConfigs}.\n */\nexport function findRootEvalConfig(rootDir: string): string | undefined {\n const file = path.join(rootDir, \"evals.config.ts\");\n return existsSync(file) ? file : undefined;\n}\n\n/**\n * Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at\n * `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:\n * each agent's config applies only to that agent's evals. Agents without a\n * config file are omitted. Returns a stable, sorted list.\n */\nexport function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEvalConfig[] = [];\n\n for (const agent of agents) {\n const file = path.join(agentsDir, agent, \"evals\", \"evals.config.ts\");\n if (existsSync(file)) out.push({ file, agent });\n }\n\n return out.sort((a, b) => a.agent.localeCompare(b.agent));\n}\n"],"mappings":";;;;;;;AAyBA,SAAS,YAAY,KAAuB;AAC1C,KAAI;AACF,SAAO,YAAY,KAAK;GAAE,WAAW;GAAM,eAAe;GAAM,CAAC,CAC9D,QAAQ,MAAM,EAAE,QAAQ,IAAI,EAAE,KAAK,SAAS,WAAW,CAAC,CACxD,KAAK,MAAM,KAAK,KAAK,EAAE,YAAY,EAAE,KAAK,CAAC;SACxC;AACN,SAAO,EAAE;;;;;;;AAQb,SAAS,WAAW,SAA0D;CAC5E,MAAM,YAAY,KAAK,KAAK,SAAS,uBAAuB;AAC5D,KAAI;AACF,SAAO;GACL;GACA,QAAQ,cAAc,YAAY,WAAW,EAAE,eAAe,MAAM,CAAC,CAAC;GACvE;SACK;AACN,SAAO;GAAE;GAAW,QAAQ,EAAE;GAAE;;;;;;;;;AAUpC,SAAgB,kBAAkB,SAAmC;CACnE,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAAwB,EAAE;AAEhC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,WAAW,KAAK,KAAK,WAAW,OAAO,QAAQ;AACrD,OAAK,MAAM,QAAQ,YAAY,SAAS,EAAE;GACxC,MAAM,KAAK,KACR,SAAS,UAAU,KAAK,CACxB,QAAQ,eAAe,GAAG,CAC1B,MAAM,KAAK,IAAI,CACf,KAAK,IAAI;AACZ,OAAI,KAAK;IAAE;IAAM;IAAI;IAAO,CAAC;;;AAIjC,QAAO,IAAI,MACR,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,IAAI,EAAE,GAAG,cAAc,EAAE,GAAG,CACrE;;;;;;;;AASH,SAAgB,mBAAmB,SAAqC;CACtE,MAAM,OAAO,KAAK,KAAK,SAAS,kBAAkB;AAClD,QAAO,WAAW,KAAK,GAAG,OAAO;;;;;;;;AASnC,SAAgB,oBAAoB,SAAyC;CAC3E,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAA8B,EAAE;AAEtC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,OAAO,KAAK,KAAK,WAAW,OAAO,SAAS,kBAAkB;AACpE,MAAI,WAAW,KAAK,CAAE,KAAI,KAAK;GAAE;GAAM;GAAO,CAAC;;AAGjD,QAAO,IAAI,MAAM,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,CAAC"}
|
package/dist/evals/index.d.ts
CHANGED
|
@@ -2,13 +2,13 @@ import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth, re
|
|
|
2
2
|
import { MlflowClient, PostResult, normalizeHost } from "../connectors/mlflow/client.js";
|
|
3
3
|
import "../connectors/mlflow/index.js";
|
|
4
4
|
import { DatasetRow, ReadEvalDatasetOptions, readEvalDataset, userTurns } from "./dataset.js";
|
|
5
|
-
import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext } from "./types.js";
|
|
5
|
+
import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, EvalWebServer, MatchResult, Matcher, Severity, TestContext } from "./types.js";
|
|
6
6
|
import { defineEval, defineEvalConfig } from "./define-eval.js";
|
|
7
|
-
import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
|
|
7
|
+
import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./discover.js";
|
|
8
8
|
import { HttpDriverOptions, createHttpDriver } from "./http-driver.js";
|
|
9
9
|
import { JudgeConfig, JudgeScore, configureJudge, isJudgeConfigured } from "./judge.js";
|
|
10
10
|
import { equals, includes, matches } from "./matchers.js";
|
|
11
11
|
import { Assessment, ReportOutcome, buildAssessments, reportToMlflow } from "./mlflow-report.js";
|
|
12
12
|
import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./report.js";
|
|
13
13
|
import { RunEvalOptions, runEval } from "./run-eval.js";
|
|
14
|
-
import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries } from "./run-evals.js";
|
|
14
|
+
import { EvalProgress, EvalRunSummary, RunEvalsOptions, loadRootEvalConfig, runEvalsInDir, runWithRetries } from "./run-evals.js";
|
package/dist/evals/index.js
CHANGED
|
@@ -2,13 +2,13 @@ import { resolveDatabricksAuth, resolveWorkspaceClient } from "../connectors/mlf
|
|
|
2
2
|
import { MlflowClient, normalizeHost } from "../connectors/mlflow/client.js";
|
|
3
3
|
import { readEvalDataset, userTurns } from "./dataset.js";
|
|
4
4
|
import { defineEval, defineEvalConfig } from "./define-eval.js";
|
|
5
|
-
import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
|
|
5
|
+
import { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./discover.js";
|
|
6
6
|
import { createHttpDriver } from "./http-driver.js";
|
|
7
7
|
import { configureJudge, isJudgeConfigured } from "./judge.js";
|
|
8
8
|
import { equals, includes, matches } from "./matchers.js";
|
|
9
9
|
import { buildAssessments, reportToMlflow } from "./mlflow-report.js";
|
|
10
10
|
import { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./report.js";
|
|
11
11
|
import { runEval } from "./run-eval.js";
|
|
12
|
-
import { runEvalsInDir, runWithRetries } from "./run-evals.js";
|
|
12
|
+
import { loadRootEvalConfig, runEvalsInDir, runWithRetries } from "./run-evals.js";
|
|
13
13
|
|
|
14
14
|
export { };
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { WorkspaceClient } from "../workspace-client/index.js";
|
|
2
2
|
import "./dataset.js";
|
|
3
|
-
import { EvalResult } from "./types.js";
|
|
3
|
+
import { EvalConfig, EvalResult } from "./types.js";
|
|
4
4
|
import { ReportOutcome } from "./mlflow-report.js";
|
|
5
5
|
import { FinishOutcome } from "./mlflow-run.js";
|
|
6
6
|
|
|
@@ -98,6 +98,13 @@ interface EvalRunSummary {
|
|
|
98
98
|
finish: FinishOutcome;
|
|
99
99
|
};
|
|
100
100
|
}
|
|
101
|
+
/**
|
|
102
|
+
* Load the root `evals.config.ts` under `rootDir` (the project root), or return
|
|
103
|
+
* `undefined` when there is none. This is the run-wide config carrying
|
|
104
|
+
* `baseUrl`/`webServer`; the CLI reads it to resolve options and manage the
|
|
105
|
+
* app-under-test lifecycle before calling {@link runEvalsInDir}.
|
|
106
|
+
*/
|
|
107
|
+
declare function loadRootEvalConfig(rootDir: string): Promise<EvalConfig | undefined>;
|
|
101
108
|
/**
|
|
102
109
|
* Run `attempt` up to `1 + retries` times, stopping as soon as it returns a
|
|
103
110
|
* result that is neither a thrown error / per-eval timeout (`error`) nor a
|
|
@@ -119,5 +126,5 @@ declare function runWithRetries(retries: number, attempt: (attemptNumber: number
|
|
|
119
126
|
*/
|
|
120
127
|
declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
|
|
121
128
|
//#endregion
|
|
122
|
-
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries };
|
|
129
|
+
export { EvalProgress, EvalRunSummary, RunEvalsOptions, loadRootEvalConfig, runEvalsInDir, runWithRetries };
|
|
123
130
|
//# sourceMappingURL=run-evals.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;
|
|
1
|
+
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAoBiB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAWU;EATV,MAAA;EAwDkB;;;;EAnDlB,IAAA;EALA;EAOA,MAAA;EAAA;EAEA,OAAA,GAAU,MAAA;EAAA;;;;;;EAOV,WAAA;EAiBA;;;;;EAXA,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UA6BF;IA3BE,cAAA;EAAA;EA6BS;;;AAGb;EA1BE,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EA4BnC;;;;EAvBJ,eAAA,GAAkB,eAAA;EAwB4B;EAtB9C,WAAA;EAuBoB;EArBpB,GAAA;EAqBwC;;;;AAE1C;EAjBE,SAAA;;;;;;EAMA,OAAA;EAYA;EAVA,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EAyPmC;EAvP5C,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;;iBAmDrC,kBAAA,CACpB,OAAA,WACC,OAAA,CAAQ,UAAA;;;;;;;;;;;;iBAgMW,cAAA,CACpB,OAAA,UACA,OAAA,GAAU,aAAA,aAA0B,OAAA,CAAQ,UAAA,GAC5C,OAAA;EAAW,WAAA;AAAA,IACV,OAAA,CAAQ,UAAA;;;;;;iBA8GW,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
|
package/dist/evals/run-evals.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
2
|
import { readEvalDataset } from "./dataset.js";
|
|
3
|
-
import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
|
|
3
|
+
import { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./discover.js";
|
|
4
4
|
import { createHttpDriver } from "./http-driver.js";
|
|
5
5
|
import { configureJudge, teardownJudge } from "./judge.js";
|
|
6
6
|
import { mapPool } from "./pool.js";
|
|
@@ -43,6 +43,17 @@ async function loadEvalConfig(file) {
|
|
|
43
43
|
return resolveConfigDefault(await tsImportFile(file));
|
|
44
44
|
}
|
|
45
45
|
/**
|
|
46
|
+
* Load the root `evals.config.ts` under `rootDir` (the project root), or return
|
|
47
|
+
* `undefined` when there is none. This is the run-wide config carrying
|
|
48
|
+
* `baseUrl`/`webServer`; the CLI reads it to resolve options and manage the
|
|
49
|
+
* app-under-test lifecycle before calling {@link runEvalsInDir}.
|
|
50
|
+
*/
|
|
51
|
+
async function loadRootEvalConfig(rootDir) {
|
|
52
|
+
const file = findRootEvalConfig(rootDir);
|
|
53
|
+
if (!file) return void 0;
|
|
54
|
+
return loadEvalConfig(file);
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
46
57
|
* Unwrap the config default export across module-interop shapes (see
|
|
47
58
|
* {@link resolveEvalDefault}). A config has no `.test`, so the first plain
|
|
48
59
|
* object reached through the `default` chain is taken as the config.
|
|
@@ -355,5 +366,5 @@ async function runEvalsInDir(options) {
|
|
|
355
366
|
}
|
|
356
367
|
|
|
357
368
|
//#endregion
|
|
358
|
-
export { runEvalsInDir, runWithRetries };
|
|
369
|
+
export { loadRootEvalConfig, runEvalsInDir, runWithRetries };
|
|
359
370
|
//# sourceMappingURL=run-evals.js.map
|