@databricks/appkit 0.72.0 → 0.74.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +24 -0
- package/NOTICE.md +1 -0
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +11 -8
- package/dist/beta.js +7 -6
- package/dist/cli/commands/agent/eval.js +85 -16
- package/dist/cli/commands/agent/eval.js.map +1 -1
- package/dist/connectors/index.js +1 -1
- package/dist/connectors/mlflow/auth.d.ts +11 -1
- package/dist/connectors/mlflow/auth.d.ts.map +1 -1
- package/dist/connectors/mlflow/auth.js +22 -2
- package/dist/connectors/mlflow/auth.js.map +1 -1
- package/dist/connectors/mlflow/index.d.ts +2 -0
- package/dist/database/errors.js +15 -5
- package/dist/database/errors.js.map +1 -1
- package/dist/database/runtime/data-path.d.ts +7 -0
- package/dist/database/runtime/data-path.d.ts.map +1 -0
- package/dist/database/runtime/data-path.js.map +1 -1
- package/dist/database/runtime/engine/drizzle-data-path.js +7 -5
- package/dist/database/runtime/engine/drizzle-data-path.js.map +1 -1
- package/dist/database/schema-builder/define-schema.d.ts +1 -1
- package/dist/database/schema-builder/define-schema.js +1 -1
- package/dist/database/schema-builder/define-schema.js.map +1 -1
- package/dist/errors/database-validation.d.ts +23 -0
- package/dist/errors/database-validation.d.ts.map +1 -0
- package/dist/errors/database-validation.js +24 -0
- package/dist/errors/database-validation.js.map +1 -0
- package/dist/errors/index.js +1 -0
- package/dist/evals/dataset.d.ts +49 -0
- package/dist/evals/dataset.d.ts.map +1 -0
- package/dist/evals/dataset.js +51 -0
- package/dist/evals/dataset.js.map +1 -0
- package/dist/evals/define-eval.d.ts +4 -2
- package/dist/evals/define-eval.d.ts.map +1 -1
- package/dist/evals/define-eval.js +5 -1
- package/dist/evals/define-eval.js.map +1 -1
- package/dist/evals/discover.d.ts +15 -1
- package/dist/evals/discover.d.ts.map +1 -1
- package/dist/evals/discover.js +26 -2
- package/dist/evals/discover.js.map +1 -1
- package/dist/evals/http-driver.d.ts.map +1 -1
- package/dist/evals/http-driver.js +82 -55
- package/dist/evals/http-driver.js.map +1 -1
- package/dist/evals/index.d.ts +14 -0
- package/dist/evals/index.js +6 -5
- package/dist/evals/judge.d.ts +1 -0
- package/dist/evals/judge.d.ts.map +1 -1
- package/dist/evals/mlflow-report.d.ts +1 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -1
- package/dist/evals/mlflow-run.d.ts +2 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -1
- package/dist/evals/report.d.ts +16 -1
- package/dist/evals/report.d.ts.map +1 -1
- package/dist/evals/report.js +64 -2
- package/dist/evals/report.js.map +1 -1
- package/dist/evals/run-eval.d.ts +8 -0
- package/dist/evals/run-eval.d.ts.map +1 -1
- package/dist/evals/run-eval.js +54 -5
- package/dist/evals/run-eval.js.map +1 -1
- package/dist/evals/run-evals.d.ts +41 -3
- package/dist/evals/run-evals.d.ts.map +1 -1
- package/dist/evals/run-evals.js +215 -34
- package/dist/evals/run-evals.js.map +1 -1
- package/dist/evals/types.d.ts +80 -6
- package/dist/evals/types.d.ts.map +1 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/plugin/plugin.d.ts.map +1 -1
- package/dist/plugin/plugin.js +1 -1
- package/dist/plugin/plugin.js.map +1 -1
- package/dist/plugins/database/crud/contract.js +17 -8
- package/dist/plugins/database/crud/contract.js.map +1 -1
- package/dist/plugins/database/crud/exposure.js +63 -22
- package/dist/plugins/database/crud/exposure.js.map +1 -1
- package/dist/plugins/database/crud/request.js +50 -0
- package/dist/plugins/database/crud/request.js.map +1 -0
- package/dist/plugins/database/crud/response.js +77 -0
- package/dist/plugins/database/crud/response.js.map +1 -0
- package/dist/plugins/database/crud/routes.js +71 -52
- package/dist/plugins/database/crud/routes.js.map +1 -1
- package/dist/plugins/database/database.d.ts +6 -4
- package/dist/plugins/database/database.d.ts.map +1 -1
- package/dist/plugins/database/database.js +46 -16
- package/dist/plugins/database/database.js.map +1 -1
- package/dist/plugins/database/defaults.js +5 -1
- package/dist/plugins/database/defaults.js.map +1 -1
- package/dist/plugins/database/entity-client.js +143 -10
- package/dist/plugins/database/entity-client.js.map +1 -1
- package/dist/plugins/database/entity-types.d.ts +1 -1
- package/dist/plugins/database/hooks.d.ts +38 -0
- package/dist/plugins/database/hooks.d.ts.map +1 -0
- package/dist/plugins/database/index.d.ts +3 -2
- package/dist/plugins/database/lifecycle.js +67 -28
- package/dist/plugins/database/lifecycle.js.map +1 -1
- package/dist/plugins/database/scope.js +58 -0
- package/dist/plugins/database/scope.js.map +1 -0
- package/dist/plugins/database/types.d.ts +40 -12
- package/dist/plugins/database/types.d.ts.map +1 -1
- package/dist/plugins/server/index.js +2 -2
- package/dist/plugins/server/index.js.map +1 -1
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js +3 -3
- package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js.map +1 -1
- package/dist/plugins/server/static-server.js +3 -3
- package/dist/plugins/server/static-server.js.map +1 -1
- package/dist/plugins/server/utils.js +3 -3
- package/dist/plugins/server/utils.js.map +1 -1
- package/dist/plugins/server/vite-dev-server.js +4 -4
- package/dist/plugins/server/vite-dev-server.js.map +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +87 -87
- package/dist/type-generator/database/generate.js +3 -3
- package/dist/type-generator/database/generate.js.map +1 -1
- package/dist/type-generator/migration.js +2 -2
- package/dist/type-generator/migration.js.map +1 -1
- package/dist/type-generator/serving/server-file-extractor.js +3 -3
- package/dist/type-generator/serving/server-file-extractor.js.map +1 -1
- package/docs/api/appkit/Class.AppKitError.md +1 -0
- package/docs/api/appkit/Class.DatabaseValidationError.md +191 -0
- package/docs/api/appkit/Function.defineEvalConfig.md +18 -0
- package/docs/api/appkit/Function.defineSchema.md +1 -1
- package/docs/api/appkit/Function.discoverEvalConfigs.md +18 -0
- package/docs/api/appkit/Function.formatResultsJUnit.md +18 -0
- package/docs/api/appkit/Function.formatResultsJson.md +18 -0
- package/docs/api/appkit/Function.readEvalDataset.md +21 -0
- package/docs/api/appkit/Function.resolveWorkspaceClient.md +18 -0
- package/docs/api/appkit/Function.runWithRetries.md +28 -0
- package/docs/api/appkit/Function.userTurns.md +20 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +1 -1
- package/docs/api/appkit/Interface.DatabaseValidationIssue.md +21 -0
- package/docs/api/appkit/Interface.DatasetRow.md +21 -0
- package/docs/api/appkit/Interface.DiscoveredEvalConfig.md +25 -0
- package/docs/api/appkit/Interface.DriveResult.md +28 -0
- package/docs/api/appkit/Interface.EntityMutationHooks.md +173 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +50 -0
- package/docs/api/appkit/Interface.EvalDriver.md +26 -5
- package/docs/api/appkit/Interface.EvalResult.md +11 -0
- package/docs/api/appkit/Interface.EvalSummary.md +11 -0
- package/docs/api/appkit/Interface.HookApp.md +12 -0
- package/docs/api/appkit/Interface.HookContext.md +21 -0
- package/docs/api/appkit/Interface.ReadEvalDatasetOptions.md +34 -0
- package/docs/api/appkit/Interface.ReadSerializerContext.md +21 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +22 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +45 -1
- package/docs/api/appkit/Interface.TestContext.md +67 -8
- package/docs/api/appkit/TypeAlias.DatabaseApiConfig.md +53 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md +8 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md +49 -0
- package/docs/api/appkit/TypeAlias.DatabaseExports.md +3 -3
- package/docs/api/appkit/TypeAlias.EntityHooks.md +25 -0
- package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +16 -5
- package/docs/api/appkit/TypeAlias.ReadSerializer.md +19 -0
- package/docs/api/appkit/TypeAlias.TransactionClient.md +19 -0
- package/docs/api/appkit.md +142 -119
- package/docs/plugins/database.md +144 -0
- package/llms.txt +24 -0
- package/package.json +2 -2
- package/sbom.cdx.json +1 -1
package/dist/evals/run-eval.js
CHANGED
|
@@ -12,6 +12,22 @@ var SkipSignal = class extends Error {
|
|
|
12
12
|
}
|
|
13
13
|
};
|
|
14
14
|
/**
|
|
15
|
+
* Deep partial match: every key in `expected` is present in `actual` and equal,
|
|
16
|
+
* recursing into nested plain objects so extra actual keys are ignored. Arrays
|
|
17
|
+
* match element-for-element (same length, deep-equal items).
|
|
18
|
+
*/
|
|
19
|
+
function deepContains(actual, expected) {
|
|
20
|
+
if (Array.isArray(expected)) return Array.isArray(actual) && actual.length === expected.length && expected.every((item, i) => deepContains(actual[i], item));
|
|
21
|
+
if (isPlainObject(expected)) {
|
|
22
|
+
if (!isPlainObject(actual)) return false;
|
|
23
|
+
return Object.keys(expected).every((key) => Object.hasOwn(actual, key) && deepContains(actual[key], expected[key]));
|
|
24
|
+
}
|
|
25
|
+
return actual === expected;
|
|
26
|
+
}
|
|
27
|
+
function isPlainObject(value) {
|
|
28
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
15
31
|
* Run a single eval against a driver. Never throws for assertion or agent
|
|
16
32
|
* failures — those become a non-passing {@link EvalResult}. Only a malformed
|
|
17
33
|
* eval definition surfaces as `result.error`.
|
|
@@ -21,9 +37,12 @@ async function runEval(def, options) {
|
|
|
21
37
|
let reply = "";
|
|
22
38
|
let lastInput = "";
|
|
23
39
|
let toolCalls = [];
|
|
40
|
+
let toolCallDetails = [];
|
|
24
41
|
let sessionId;
|
|
25
42
|
let lastTraceId;
|
|
26
43
|
let lastSucceeded = false;
|
|
44
|
+
let turnFailed = false;
|
|
45
|
+
const controller = new AbortController();
|
|
27
46
|
const record = (label, pass, score, detail) => {
|
|
28
47
|
const result = {
|
|
29
48
|
label,
|
|
@@ -43,24 +62,28 @@ async function runEval(def, options) {
|
|
|
43
62
|
return handle;
|
|
44
63
|
},
|
|
45
64
|
atLeast(threshold) {
|
|
46
|
-
result.severity = "soft";
|
|
47
65
|
result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;
|
|
48
66
|
return handle;
|
|
49
67
|
}
|
|
50
68
|
};
|
|
51
69
|
return handle;
|
|
52
70
|
};
|
|
53
|
-
const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale)
|
|
71
|
+
const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);
|
|
54
72
|
const t = {
|
|
55
73
|
async send(message) {
|
|
56
74
|
lastInput = message;
|
|
57
|
-
const r = await options.driver.send(message);
|
|
75
|
+
const r = await options.driver.send(message, { signal: controller.signal });
|
|
58
76
|
reply = r.reply;
|
|
59
77
|
toolCalls = r.toolCalls;
|
|
78
|
+
toolCallDetails = r.toolCallDetails;
|
|
60
79
|
sessionId = r.sessionId;
|
|
61
80
|
lastSucceeded = r.succeeded;
|
|
81
|
+
if (!r.succeeded) turnFailed = true;
|
|
62
82
|
if (r.traceId) lastTraceId = r.traceId;
|
|
63
83
|
},
|
|
84
|
+
reset() {
|
|
85
|
+
options.driver.reset?.();
|
|
86
|
+
},
|
|
64
87
|
get reply() {
|
|
65
88
|
return reply;
|
|
66
89
|
},
|
|
@@ -70,12 +93,24 @@ async function runEval(def, options) {
|
|
|
70
93
|
get sessionId() {
|
|
71
94
|
return sessionId;
|
|
72
95
|
},
|
|
96
|
+
get input() {
|
|
97
|
+
return options.row?.inputs ?? {};
|
|
98
|
+
},
|
|
99
|
+
get expected() {
|
|
100
|
+
return options.row?.expectations;
|
|
101
|
+
},
|
|
73
102
|
succeeded() {
|
|
74
103
|
return record("succeeded", lastSucceeded, void 0, lastSucceeded ? void 0 : "agent turn did not complete successfully");
|
|
75
104
|
},
|
|
76
105
|
calledTool(name) {
|
|
77
106
|
return record(`calledTool(${name})`, toolCalls.includes(name), void 0, `expected tool "${name}" to be called (called: ${toolCalls.length ? toolCalls.join(", ") : "none"})`);
|
|
78
107
|
},
|
|
108
|
+
calledToolWith(name, expected) {
|
|
109
|
+
const matching = toolCallDetails.filter((c) => c.name === name);
|
|
110
|
+
const pass = matching.some((c) => deepContains(c.args, expected));
|
|
111
|
+
const seen = matching.length ? matching.map((c) => `{${Object.keys(c.args).sort().join(", ")}}`).join(", ") : "not called";
|
|
112
|
+
return record(`calledToolWith(${name})`, pass, void 0, `expected tool "${name}" to be called with ${JSON.stringify(expected)} (arg keys seen: ${seen})`);
|
|
113
|
+
},
|
|
79
114
|
check(value, matcher) {
|
|
80
115
|
const m = matcher(value);
|
|
81
116
|
return record("check", m.pass, m.score, m.detail);
|
|
@@ -109,8 +144,19 @@ async function runEval(def, options) {
|
|
|
109
144
|
throw new SkipSignal(reason);
|
|
110
145
|
}
|
|
111
146
|
};
|
|
147
|
+
const timeoutMs = def.timeoutMs ?? options.timeoutMs;
|
|
148
|
+
let timer;
|
|
112
149
|
try {
|
|
113
|
-
await def.test(t);
|
|
150
|
+
if (timeoutMs === void 0) await def.test(t);
|
|
151
|
+
else {
|
|
152
|
+
const timeout = new Promise((_, reject) => {
|
|
153
|
+
timer = setTimeout(() => {
|
|
154
|
+
controller.abort();
|
|
155
|
+
reject(/* @__PURE__ */ new Error(`eval timed out after ${timeoutMs}ms`));
|
|
156
|
+
}, timeoutMs);
|
|
157
|
+
});
|
|
158
|
+
await Promise.race([Promise.resolve(def.test(t)), timeout]);
|
|
159
|
+
}
|
|
114
160
|
} catch (err) {
|
|
115
161
|
if (err instanceof SkipSignal) return {
|
|
116
162
|
id: options.id,
|
|
@@ -128,6 +174,8 @@ async function runEval(def, options) {
|
|
|
128
174
|
error: err instanceof Error ? err.message : String(err),
|
|
129
175
|
traceId: lastTraceId
|
|
130
176
|
};
|
|
177
|
+
} finally {
|
|
178
|
+
if (timer) clearTimeout(timer);
|
|
131
179
|
}
|
|
132
180
|
const passed = assertions.every((a) => a.pass || a.severity === "soft" && !options.strict);
|
|
133
181
|
return {
|
|
@@ -135,7 +183,8 @@ async function runEval(def, options) {
|
|
|
135
183
|
description: def.description,
|
|
136
184
|
assertions,
|
|
137
185
|
passed,
|
|
138
|
-
traceId: lastTraceId
|
|
186
|
+
traceId: lastTraceId,
|
|
187
|
+
infraFailure: turnFailed && !passed || void 0
|
|
139
188
|
};
|
|
140
189
|
}
|
|
141
190
|
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n result.severity = \"soft\";\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and soft by default; the caller chains\n // `.atLeast(n)` to set the pass threshold or `.gate()` to promote.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale).soft();\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message);\n reply = r.reply;\n toolCalls = r.toolCalls;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (r.traceId) lastTraceId = r.traceId;\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n try {\n await def.test(t);\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n };\n}\n"],"mappings":";;;;AAYA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAkBhB,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AACzB,WAAO,WAAW;AAClB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAKT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU,CAAC,MAAM;CAE1E,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,QAAQ;AAC5C,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;AAED,KAAI;AACF,QAAM,IAAI,KAAK,EAAE;UACV,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;;CAGH,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EACV"}
|
|
1
|
+
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import type { DatasetRow } from \"./dataset\";\nimport { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n DriveResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\n/**\n * Deep partial match: every key in `expected` is present in `actual` and equal,\n * recursing into nested plain objects so extra actual keys are ignored. Arrays\n * match element-for-element (same length, deep-equal items).\n */\nfunction deepContains(actual: unknown, expected: unknown): boolean {\n if (Array.isArray(expected)) {\n return (\n Array.isArray(actual) &&\n actual.length === expected.length &&\n expected.every((item, i) => deepContains(actual[i], item))\n );\n }\n if (isPlainObject(expected)) {\n if (!isPlainObject(actual)) return false;\n // Require the key present: an expected `undefined` must not match an omitted key.\n return Object.keys(expected).every(\n (key) =>\n Object.hasOwn(actual, key) && deepContains(actual[key], expected[key]),\n );\n }\n return actual === expected;\n}\n\nfunction isPlainObject(value: unknown): value is Record<string, unknown> {\n return typeof value === \"object\" && value !== null && !Array.isArray(value);\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */\n row?: DatasetRow;\n /**\n * Runner-level default per-eval timeout (ms). `def.timeoutMs` wins over this;\n * when both are unset the eval runs unbounded (current behavior).\n */\n timeoutMs?: number;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let toolCallDetails: DriveResult[\"toolCallDetails\"] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n // Any transport/agent turn failure (driver `succeeded: false`) → `infraFailure`, for retry.\n let turnFailed = false;\n // Aborted on per-eval timeout to cancel the in-flight turn (no leaked stream).\n const controller = new AbortController();\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n // Re-threshold the score; keep the current severity (gate unless the\n // caller also chained `.soft()`).\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and gate by default (a miss fails the eval,\n // like other assertions). The caller chains `.atLeast(n)` to change the pass\n // threshold, or `.soft()` to demote to a tracked-only metric.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message, {\n signal: controller.signal,\n });\n reply = r.reply;\n toolCalls = r.toolCalls;\n toolCallDetails = r.toolCallDetails;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (!r.succeeded) turnFailed = true;\n if (r.traceId) lastTraceId = r.traceId;\n },\n reset() {\n options.driver.reset?.();\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n get input() {\n return options.row?.inputs ?? {};\n },\n get expected() {\n return options.row?.expectations;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n calledToolWith(name, expected) {\n const matching = toolCallDetails.filter((c) => c.name === name);\n const pass = matching.some((c) => deepContains(c.args, expected));\n // Keys only, never values: actual args may hold PII/secrets and are persisted to reports (CWE-532).\n const seen = matching.length\n ? matching\n .map((c) => `{${Object.keys(c.args).sort().join(\", \")}}`)\n .join(\", \")\n : \"not called\";\n return record(\n `calledToolWith(${name})`,\n pass,\n undefined,\n `expected tool \"${name}\" to be called with ${JSON.stringify(\n expected,\n )} (arg keys seen: ${seen})`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n // `def.timeoutMs` (per-eval) wins over the runner default; when both are\n // unset the eval runs unbounded (undefined = no timeout).\n const timeoutMs = def.timeoutMs ?? options.timeoutMs;\n let timer: ReturnType<typeof setTimeout> | undefined;\n\n try {\n if (timeoutMs === undefined) {\n await def.test(t);\n } else {\n // Race the test against the timeout; on elapse, abort the turn and settle\n // non-passing. Only the driver turn cancels — non-driver hangs (judge, sleep) run on.\n const timeout = new Promise<never>((_, reject) => {\n timer = setTimeout(() => {\n controller.abort();\n reject(new Error(`eval timed out after ${timeoutMs}ms`));\n }, timeoutMs);\n });\n await Promise.race([Promise.resolve(def.test(t)), timeout]);\n }\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n } finally {\n if (timer) clearTimeout(timer);\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n // Only a *failing* eval whose turn broke at the transport/agent level is a\n // retryable infra flake; a passing eval (or a pure assertion mismatch, where\n // the turn itself succeeded) is real signal and must not be retried.\n infraFailure: (turnFailed && !passed) || undefined,\n };\n}\n"],"mappings":";;;;AAcA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAShB,SAAS,aAAa,QAAiB,UAA4B;AACjE,KAAI,MAAM,QAAQ,SAAS,CACzB,QACE,MAAM,QAAQ,OAAO,IACrB,OAAO,WAAW,SAAS,UAC3B,SAAS,OAAO,MAAM,MAAM,aAAa,OAAO,IAAI,KAAK,CAAC;AAG9D,KAAI,cAAc,SAAS,EAAE;AAC3B,MAAI,CAAC,cAAc,OAAO,CAAE,QAAO;AAEnC,SAAO,OAAO,KAAK,SAAS,CAAC,OAC1B,QACC,OAAO,OAAO,QAAQ,IAAI,IAAI,aAAa,OAAO,MAAM,SAAS,KAAK,CACzE;;AAEH,QAAO,WAAW;;AAGpB,SAAS,cAAc,OAAkD;AACvE,QAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,MAAM;;;;;;;AAwB7E,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI,kBAAkD,EAAE;CACxD,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,IAAI,aAAa;CAEjB,MAAM,aAAa,IAAI,iBAAiB;CAExC,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AAGzB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAMT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU;CAEnE,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,SAAS,EAC3C,QAAQ,WAAW,QACpB,CAAC;AACF,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,qBAAkB,EAAE;AACpB,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,CAAC,EAAE,UAAW,cAAa;AAC/B,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,QAAQ;AACN,WAAQ,OAAO,SAAS;;EAE1B,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,QAAQ;AACV,UAAO,QAAQ,KAAK,UAAU,EAAE;;EAElC,IAAI,WAAW;AACb,UAAO,QAAQ,KAAK;;EAEtB,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,eAAe,MAAM,UAAU;GAC7B,MAAM,WAAW,gBAAgB,QAAQ,MAAM,EAAE,SAAS,KAAK;GAC/D,MAAM,OAAO,SAAS,MAAM,MAAM,aAAa,EAAE,MAAM,SAAS,CAAC;GAEjE,MAAM,OAAO,SAAS,SAClB,SACG,KAAK,MAAM,IAAI,OAAO,KAAK,EAAE,KAAK,CAAC,MAAM,CAAC,KAAK,KAAK,CAAC,GAAG,CACxD,KAAK,KAAK,GACb;AACJ,UAAO,OACL,kBAAkB,KAAK,IACvB,MACA,QACA,kBAAkB,KAAK,sBAAsB,KAAK,UAChD,SACD,CAAC,mBAAmB,KAAK,GAC3B;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;CAID,MAAM,YAAY,IAAI,aAAa,QAAQ;CAC3C,IAAI;AAEJ,KAAI;AACF,MAAI,cAAc,OAChB,OAAM,IAAI,KAAK,EAAE;OACZ;GAGL,MAAM,UAAU,IAAI,SAAgB,GAAG,WAAW;AAChD,YAAQ,iBAAiB;AACvB,gBAAW,OAAO;AAClB,4BAAO,IAAI,MAAM,wBAAwB,UAAU,IAAI,CAAC;OACvD,UAAU;KACb;AACF,SAAM,QAAQ,KAAK,CAAC,QAAQ,QAAQ,IAAI,KAAK,EAAE,CAAC,EAAE,QAAQ,CAAC;;UAEtD,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;WACO;AACR,MAAI,MAAO,cAAa,MAAM;;CAGhC,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EAIT,cAAe,cAAc,CAAC,UAAW;EAC1C"}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { WorkspaceClient } from "../workspace-client/index.js";
|
|
2
|
+
import "./dataset.js";
|
|
1
3
|
import { EvalResult } from "./types.js";
|
|
2
4
|
import { ReportOutcome } from "./mlflow-report.js";
|
|
3
5
|
import { FinishOutcome } from "./mlflow-run.js";
|
|
@@ -10,12 +12,15 @@ interface RunEvalsOptions {
|
|
|
10
12
|
baseUrl: string;
|
|
11
13
|
/** Substring filter on `<agent>/<id>` (or an exact agent id). */
|
|
12
14
|
filter?: string;
|
|
15
|
+
/**
|
|
16
|
+
* Only run evals whose `tags` intersect this list. Empty/undefined runs all.
|
|
17
|
+
* Tags live on the eval def, so filtering happens after each file is loaded.
|
|
18
|
+
*/
|
|
19
|
+
tags?: string[];
|
|
13
20
|
/** Soft assertion failures also fail the eval. */
|
|
14
21
|
strict?: boolean;
|
|
15
22
|
/** Extra request headers for the driver (e.g. auth for a deployed app). */
|
|
16
23
|
headers?: Record<string, string>;
|
|
17
|
-
/** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */
|
|
18
|
-
timeoutMs?: number;
|
|
19
24
|
/**
|
|
20
25
|
* Max evals to drive concurrently. Each eval opens one stream to the app as
|
|
21
26
|
* the same user, so keep this at or below the app's
|
|
@@ -43,8 +48,27 @@ interface RunEvalsOptions {
|
|
|
43
48
|
token: string;
|
|
44
49
|
model: string;
|
|
45
50
|
};
|
|
51
|
+
/**
|
|
52
|
+
* Workspace client used to read managed evaluation datasets (for evals that
|
|
53
|
+
* declare `dataset`). Required alongside {@link warehouseId} for those evals.
|
|
54
|
+
*/
|
|
55
|
+
workspaceClient?: WorkspaceClient;
|
|
56
|
+
/** SQL warehouse id used to read managed evaluation datasets. */
|
|
57
|
+
warehouseId?: string;
|
|
46
58
|
/** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */
|
|
47
59
|
now?: number;
|
|
60
|
+
/**
|
|
61
|
+
* Default per-eval timeout (ms): `runEval` races the whole test against it and
|
|
62
|
+
* it also caps each driver turn. A per-eval `def.timeoutMs` overrides it, and
|
|
63
|
+
* it wins over an agent's `evals.config.ts` `timeoutMs`. Unbounded when unset.
|
|
64
|
+
*/
|
|
65
|
+
timeoutMs?: number;
|
|
66
|
+
/**
|
|
67
|
+
* Re-run an eval up to this many extra times when it fails on infrastructure —
|
|
68
|
+
* a thrown error/timeout (`result.error`) or a transport/agent turn failure
|
|
69
|
+
* (`result.infraFailure`). Assertion failures are never retried. Defaults to `0`.
|
|
70
|
+
*/
|
|
71
|
+
retries?: number;
|
|
48
72
|
/** Progress callback, invoked as evals are discovered, started, and finished. */
|
|
49
73
|
onEvent?: (event: EvalProgress) => void;
|
|
50
74
|
}
|
|
@@ -74,6 +98,20 @@ interface EvalRunSummary {
|
|
|
74
98
|
finish: FinishOutcome;
|
|
75
99
|
};
|
|
76
100
|
}
|
|
101
|
+
/**
|
|
102
|
+
* Run `attempt` up to `1 + retries` times, stopping as soon as it returns a
|
|
103
|
+
* result that is neither a thrown error / per-eval timeout (`error`) nor a
|
|
104
|
+
* transport/agent turn failure (`infraFailure`). Assertion failures set
|
|
105
|
+
* neither, so a failed-but-completed eval is returned on the first try and
|
|
106
|
+
* never retried. Returns the last result when every attempt failed on infra.
|
|
107
|
+
*
|
|
108
|
+
* Between attempts it waits a full-jittered exponential backoff (infra flakes
|
|
109
|
+
* are overload-correlated). `retries` is coerced to a finite non-negative
|
|
110
|
+
* integer; `baseDelayMs: 0` disables the wait (tests).
|
|
111
|
+
*/
|
|
112
|
+
declare function runWithRetries(retries: number, attempt: (attemptNumber: number) => Promise<EvalResult>, options?: {
|
|
113
|
+
baseDelayMs?: number;
|
|
114
|
+
}): Promise<EvalResult>;
|
|
77
115
|
/**
|
|
78
116
|
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
79
117
|
* the agents on a running app. Never throws for an individual eval — load/run
|
|
@@ -81,5 +119,5 @@ interface EvalRunSummary {
|
|
|
81
119
|
*/
|
|
82
120
|
declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
|
|
83
121
|
//#endregion
|
|
84
|
-
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir };
|
|
122
|
+
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries };
|
|
85
123
|
//# sourceMappingURL=run-evals.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAmBiB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAWU;EATV,MAAA;EAwDkB;;;;EAnDlB,IAAA;EALA;EAOA,MAAA;EAAA;EAEA,OAAA,GAAU,MAAA;EAAA;;;;;;EAOV,WAAA;EAiBA;;;;;EAXA,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UA6BF;IA3BE,cAAA;EAAA;EA6BS;;;AAGb;EA1BE,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EA4BnC;;;;EAvBJ,eAAA,GAAkB,eAAA;EAwB4B;EAtB9C,WAAA;EAuBoB;EArBpB,GAAA;EAqBwC;;;;AAE1C;EAjBE,SAAA;;;;;;EAMA,OAAA;EAYA;EAVA,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EA2OmC;EAzO5C,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;;;;;;;iBAuOrC,cAAA,CACpB,OAAA,UACA,OAAA,GAAU,aAAA,aAA0B,OAAA,CAAQ,UAAA,GAC5C,OAAA;EAAW,WAAA;AAAA,IACV,OAAA,CAAQ,UAAA;;;;;;iBA8GW,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
|
package/dist/evals/run-evals.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
|
-
import {
|
|
2
|
+
import { readEvalDataset } from "./dataset.js";
|
|
3
|
+
import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
|
|
3
4
|
import { createHttpDriver } from "./http-driver.js";
|
|
4
5
|
import { configureJudge, teardownJudge } from "./judge.js";
|
|
5
6
|
import { mapPool } from "./pool.js";
|
|
@@ -7,15 +8,15 @@ import { reportToMlflow } from "./mlflow-report.js";
|
|
|
7
8
|
import { runEval } from "./run-eval.js";
|
|
8
9
|
import { createEvalRun, finishEvalRun } from "./mlflow-run.js";
|
|
9
10
|
import { pathToFileURL } from "node:url";
|
|
11
|
+
import { setTimeout } from "node:timers/promises";
|
|
10
12
|
|
|
11
13
|
//#region src/evals/run-evals.ts
|
|
12
14
|
/**
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* tsx's internal entry.
|
|
15
|
+
* Import a TypeScript file with tsx's programmatic loader so eval files run
|
|
16
|
+
* without a build step. The specifier is indirected so the type checker doesn't
|
|
17
|
+
* try to resolve tsx's internal entry.
|
|
17
18
|
*/
|
|
18
|
-
async function
|
|
19
|
+
async function tsImportFile(file) {
|
|
19
20
|
const tsxApi = "tsx/esm/api";
|
|
20
21
|
let tsImport;
|
|
21
22
|
try {
|
|
@@ -23,11 +24,38 @@ async function loadEval(file) {
|
|
|
23
24
|
} catch {
|
|
24
25
|
throw new Error("Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).");
|
|
25
26
|
}
|
|
26
|
-
|
|
27
|
+
return tsImport(pathToFileURL(file).href, import.meta.url);
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.
|
|
31
|
+
*/
|
|
32
|
+
async function loadEval(file) {
|
|
33
|
+
const def = resolveEvalDefault(await tsImportFile(file));
|
|
27
34
|
if (!def) throw new Error(`${file}: must default-export defineEval({ test })`);
|
|
28
35
|
return def;
|
|
29
36
|
}
|
|
30
37
|
/**
|
|
38
|
+
* Load an `evals.config.ts` file and return its default-exported
|
|
39
|
+
* {@link EvalConfig}. A malformed/missing default surfaces as `undefined` so a
|
|
40
|
+
* bad config never aborts a whole run.
|
|
41
|
+
*/
|
|
42
|
+
async function loadEvalConfig(file) {
|
|
43
|
+
return resolveConfigDefault(await tsImportFile(file));
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Unwrap the config default export across module-interop shapes (see
|
|
47
|
+
* {@link resolveEvalDefault}). A config has no `.test`, so the first plain
|
|
48
|
+
* object reached through the `default` chain is taken as the config.
|
|
49
|
+
*/
|
|
50
|
+
function resolveConfigDefault(mod) {
|
|
51
|
+
let candidate = mod;
|
|
52
|
+
for (let i = 0; i < 4 && candidate; i++) {
|
|
53
|
+
const next = candidate.default;
|
|
54
|
+
if (next === void 0) return typeof candidate === "object" ? candidate : void 0;
|
|
55
|
+
candidate = next;
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
31
59
|
* Unwrap the eval default export across module-interop shapes. Depending on
|
|
32
60
|
* whether the eval file is treated as ESM or CJS, the value lands at
|
|
33
61
|
* `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or
|
|
@@ -41,23 +69,26 @@ function resolveEvalDefault(mod) {
|
|
|
41
69
|
}
|
|
42
70
|
}
|
|
43
71
|
/**
|
|
44
|
-
*
|
|
45
|
-
*
|
|
72
|
+
* Run one eval turn against a fresh driver. Never throws — a run failure becomes
|
|
73
|
+
* a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`
|
|
74
|
+
* binds the current managed-dataset row (see {@link resolveDatasetRows}), or is
|
|
75
|
+
* `undefined` for a plain single-run eval.
|
|
46
76
|
*/
|
|
47
|
-
async function runOne(d, id, runId, options) {
|
|
77
|
+
async function runOne(d, id, def, row, runId, options) {
|
|
48
78
|
try {
|
|
49
|
-
|
|
50
|
-
return await runEval(def, {
|
|
79
|
+
return await runWithRetries(options.retries ?? 0, () => runEval(def, {
|
|
51
80
|
id,
|
|
52
81
|
driver: createHttpDriver({
|
|
53
82
|
baseUrl: options.baseUrl,
|
|
54
83
|
agent: def.agent ?? d.agent,
|
|
55
84
|
headers: options.headers,
|
|
56
85
|
mlflowRunId: runId,
|
|
57
|
-
timeoutMs: options.timeoutMs
|
|
86
|
+
timeoutMs: def.timeoutMs ?? options.timeoutMs
|
|
58
87
|
}),
|
|
59
|
-
strict: options.strict
|
|
60
|
-
|
|
88
|
+
strict: options.strict,
|
|
89
|
+
row,
|
|
90
|
+
timeoutMs: options.timeoutMs
|
|
91
|
+
}));
|
|
61
92
|
} catch (err) {
|
|
62
93
|
return {
|
|
63
94
|
id,
|
|
@@ -67,6 +98,129 @@ async function runOne(d, id, runId, options) {
|
|
|
67
98
|
};
|
|
68
99
|
}
|
|
69
100
|
}
|
|
101
|
+
/**
|
|
102
|
+
* Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval
|
|
103
|
+
* yields a single `undefined` row; a dataset eval reads its Unity Catalog table
|
|
104
|
+
* via {@link readEvalDataset}. On misconfiguration or read failure, returns a
|
|
105
|
+
* single `undefined` row plus an `error`, so the eval still surfaces one result.
|
|
106
|
+
*/
|
|
107
|
+
async function resolveDatasetRows(def, options) {
|
|
108
|
+
if (!def.dataset) return { rows: [void 0] };
|
|
109
|
+
if (!options.workspaceClient || !options.warehouseId) return {
|
|
110
|
+
rows: [void 0],
|
|
111
|
+
error: "dataset eval requires a workspace client and warehouse (pass --warehouse-id)"
|
|
112
|
+
};
|
|
113
|
+
try {
|
|
114
|
+
const rows = await readEvalDataset(options.workspaceClient, {
|
|
115
|
+
table: def.dataset.table,
|
|
116
|
+
warehouseId: options.warehouseId,
|
|
117
|
+
limit: def.dataset.limit
|
|
118
|
+
});
|
|
119
|
+
if (rows.length === 0) return {
|
|
120
|
+
rows: [void 0],
|
|
121
|
+
error: `dataset "${def.dataset.table}" returned no rows`
|
|
122
|
+
};
|
|
123
|
+
return { rows };
|
|
124
|
+
} catch (err) {
|
|
125
|
+
return {
|
|
126
|
+
rows: [void 0],
|
|
127
|
+
error: err instanceof Error ? err.message : String(err)
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* Run one already-loaded eval (from the `loaded` pre-pass), expanding a
|
|
133
|
+
* dataset-driven eval into one run per row. Appends one result per row to
|
|
134
|
+
* `results`, emitting `start`/`result` around each. Never throws: a load error
|
|
135
|
+
* (carried in `loadError`) or a dataset-read failure surfaces as a non-passing
|
|
136
|
+
* result. `total` counts eval files, not rows — per-row detail is carried in the
|
|
137
|
+
* result id (`[row i/n]`).
|
|
138
|
+
*/
|
|
139
|
+
async function runDiscovered(d, def, loadError, index, total, runId, options, emit, results) {
|
|
140
|
+
const id = `${d.agent}/${d.id}`;
|
|
141
|
+
if (loadError) {
|
|
142
|
+
emit({
|
|
143
|
+
type: "start",
|
|
144
|
+
id,
|
|
145
|
+
index,
|
|
146
|
+
total
|
|
147
|
+
});
|
|
148
|
+
const result = {
|
|
149
|
+
id,
|
|
150
|
+
assertions: [],
|
|
151
|
+
passed: false,
|
|
152
|
+
error: loadError
|
|
153
|
+
};
|
|
154
|
+
results.push(result);
|
|
155
|
+
emit({
|
|
156
|
+
type: "result",
|
|
157
|
+
result,
|
|
158
|
+
index,
|
|
159
|
+
total
|
|
160
|
+
});
|
|
161
|
+
return;
|
|
162
|
+
}
|
|
163
|
+
const { rows, error: datasetError } = await resolveDatasetRows(def, options);
|
|
164
|
+
for (let r = 0; r < rows.length; r++) {
|
|
165
|
+
const rowId = def.dataset && rows.length > 1 ? `${id} [row ${r + 1}/${rows.length}]` : id;
|
|
166
|
+
emit({
|
|
167
|
+
type: "start",
|
|
168
|
+
id: rowId,
|
|
169
|
+
index,
|
|
170
|
+
total
|
|
171
|
+
});
|
|
172
|
+
const result = datasetError ? {
|
|
173
|
+
id: rowId,
|
|
174
|
+
assertions: [],
|
|
175
|
+
passed: false,
|
|
176
|
+
error: datasetError
|
|
177
|
+
} : await runOne(d, rowId, def, rows[r], runId, options);
|
|
178
|
+
results.push(result);
|
|
179
|
+
emit({
|
|
180
|
+
type: "result",
|
|
181
|
+
result,
|
|
182
|
+
index,
|
|
183
|
+
total
|
|
184
|
+
});
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
/** Base delay (ms) before the first retry; doubled per attempt, full-jittered, capped. */
|
|
188
|
+
const DEFAULT_RETRY_BASE_DELAY_MS = 250;
|
|
189
|
+
/** Ceiling for a single retry backoff wait (ms). */
|
|
190
|
+
const MAX_RETRY_DELAY_MS = 5e3;
|
|
191
|
+
/**
|
|
192
|
+
* Run `attempt` up to `1 + retries` times, stopping as soon as it returns a
|
|
193
|
+
* result that is neither a thrown error / per-eval timeout (`error`) nor a
|
|
194
|
+
* transport/agent turn failure (`infraFailure`). Assertion failures set
|
|
195
|
+
* neither, so a failed-but-completed eval is returned on the first try and
|
|
196
|
+
* never retried. Returns the last result when every attempt failed on infra.
|
|
197
|
+
*
|
|
198
|
+
* Between attempts it waits a full-jittered exponential backoff (infra flakes
|
|
199
|
+
* are overload-correlated). `retries` is coerced to a finite non-negative
|
|
200
|
+
* integer; `baseDelayMs: 0` disables the wait (tests).
|
|
201
|
+
*/
|
|
202
|
+
async function runWithRetries(retries, attempt, options = {}) {
|
|
203
|
+
const baseDelayMs = options.baseDelayMs ?? DEFAULT_RETRY_BASE_DELAY_MS;
|
|
204
|
+
const maxAttempts = 1 + (Number.isFinite(retries) ? Math.max(0, Math.floor(retries)) : 0);
|
|
205
|
+
let result;
|
|
206
|
+
for (let n = 1;; n++) {
|
|
207
|
+
result = await attempt(n);
|
|
208
|
+
if (!(result.error !== void 0 || result.infraFailure) || n >= maxAttempts) return result;
|
|
209
|
+
if (baseDelayMs > 0) {
|
|
210
|
+
const ceiling = Math.min(baseDelayMs * 2 ** (n - 1), MAX_RETRY_DELAY_MS);
|
|
211
|
+
await setTimeout(Math.random() * ceiling);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
/**
|
|
216
|
+
* Whether an eval's `tags` satisfy a `--tag` filter: `true` when the filter is
|
|
217
|
+
* empty/undefined (no filtering), otherwise only when the eval shares at least
|
|
218
|
+
* one tag with it. An eval with no tags never matches a non-empty filter.
|
|
219
|
+
*/
|
|
220
|
+
function matchesTags(defTags, filterTags) {
|
|
221
|
+
if (!filterTags || filterTags.length === 0) return true;
|
|
222
|
+
return defTags?.some((t) => filterTags.includes(t)) ?? false;
|
|
223
|
+
}
|
|
70
224
|
/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */
|
|
71
225
|
async function maybeConfigureJudge(options) {
|
|
72
226
|
if (!options.judge) return;
|
|
@@ -113,6 +267,16 @@ async function finalizeMlflow(client, runId, results, options) {
|
|
|
113
267
|
*/
|
|
114
268
|
const DEFAULT_CONCURRENCY = 4;
|
|
115
269
|
/**
|
|
270
|
+
* Resolve the work-pool width: `--concurrency` wins; else the lowest
|
|
271
|
+
* `maxConcurrency` any *participating* agent's `evals.config.ts` requests (all
|
|
272
|
+
* evals share one per-user stream budget, so the most conservative ceiling
|
|
273
|
+
* governs); else {@link DEFAULT_CONCURRENCY}.
|
|
274
|
+
*/
|
|
275
|
+
function deriveConcurrency(activeAgents, configs, cliConcurrency) {
|
|
276
|
+
const configMin = [...configs.entries()].filter(([agent]) => activeAgents.has(agent)).map(([, c]) => c.maxConcurrency).filter((n) => typeof n === "number").reduce((min, n) => min === void 0 ? n : Math.min(min, n), void 0);
|
|
277
|
+
return cliConcurrency ?? configMin ?? DEFAULT_CONCURRENCY;
|
|
278
|
+
}
|
|
279
|
+
/**
|
|
116
280
|
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
117
281
|
* the agents on a running app. Never throws for an individual eval — load/run
|
|
118
282
|
* failures become non-passing {@link EvalResult}s.
|
|
@@ -126,7 +290,32 @@ async function runEvalsInDir(options) {
|
|
|
126
290
|
discovered = discovered.filter((d) => d.agent === f || `${d.agent}/${d.id}`.includes(f));
|
|
127
291
|
}
|
|
128
292
|
const emit = options.onEvent ?? (() => {});
|
|
129
|
-
const
|
|
293
|
+
const configs = /* @__PURE__ */ new Map();
|
|
294
|
+
for (const c of discoverEvalConfigs(root)) try {
|
|
295
|
+
const cfg = await loadEvalConfig(c.file);
|
|
296
|
+
if (cfg) configs.set(c.agent, cfg);
|
|
297
|
+
} catch {}
|
|
298
|
+
const loaded = [];
|
|
299
|
+
for (const d of discovered) {
|
|
300
|
+
let def;
|
|
301
|
+
try {
|
|
302
|
+
def = await loadEval(d.file);
|
|
303
|
+
} catch (err) {
|
|
304
|
+
loaded.push({
|
|
305
|
+
d,
|
|
306
|
+
def: { test: () => {} },
|
|
307
|
+
loadError: err instanceof Error ? err.message : String(err)
|
|
308
|
+
});
|
|
309
|
+
continue;
|
|
310
|
+
}
|
|
311
|
+
if (!matchesTags(def.tags, options.tags)) continue;
|
|
312
|
+
loaded.push({
|
|
313
|
+
d,
|
|
314
|
+
def
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
const concurrency = deriveConcurrency(new Set(loaded.map((l) => l.d.agent)), configs, options.concurrency);
|
|
318
|
+
const total = loaded.length;
|
|
130
319
|
emit({
|
|
131
320
|
type: "discovered",
|
|
132
321
|
total
|
|
@@ -147,23 +336,15 @@ async function runEvalsInDir(options) {
|
|
|
147
336
|
runId
|
|
148
337
|
});
|
|
149
338
|
}
|
|
150
|
-
const results = await mapPool(
|
|
151
|
-
const
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
emit({
|
|
160
|
-
type: "result",
|
|
161
|
-
result,
|
|
162
|
-
index,
|
|
163
|
-
total
|
|
164
|
-
});
|
|
165
|
-
return result;
|
|
166
|
-
});
|
|
339
|
+
const results = (await mapPool(loaded, concurrency, async ({ d, def, loadError }, index) => {
|
|
340
|
+
const fileResults = [];
|
|
341
|
+
const fileOptions = {
|
|
342
|
+
...options,
|
|
343
|
+
timeoutMs: options.timeoutMs ?? configs.get(d.agent)?.timeoutMs
|
|
344
|
+
};
|
|
345
|
+
await runDiscovered(d, def, loadError, index, total, runId, fileOptions, emit, fileResults);
|
|
346
|
+
return fileResults;
|
|
347
|
+
})).flat();
|
|
167
348
|
const summary = { results };
|
|
168
349
|
const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);
|
|
169
350
|
if (mlflow) summary.mlflow = mlflow;
|
|
@@ -174,5 +355,5 @@ async function runEvalsInDir(options) {
|
|
|
174
355
|
}
|
|
175
356
|
|
|
176
357
|
//#endregion
|
|
177
|
-
export { runEvalsInDir };
|
|
358
|
+
export { runEvalsInDir, runWithRetries };
|
|
178
359
|
//# sourceMappingURL=run-evals.js.map
|