@databricks/appkit 0.71.0 → 0.72.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +46 -0
- package/NOTICE.md +2 -2
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +13 -1
- package/dist/beta.js +13 -1
- package/dist/cli/commands/agent/eval.js +112 -0
- package/dist/cli/commands/agent/eval.js.map +1 -0
- package/dist/cli/commands/agent/index.js +18 -0
- package/dist/cli/commands/agent/index.js.map +1 -0
- package/dist/cli/index.js +2 -0
- package/dist/cli/index.js.map +1 -1
- package/dist/connectors/index.js +2 -0
- package/dist/connectors/mlflow/auth.d.ts +18 -0
- package/dist/connectors/mlflow/auth.d.ts.map +1 -0
- package/dist/connectors/mlflow/auth.js +50 -0
- package/dist/connectors/mlflow/auth.js.map +1 -0
- package/dist/connectors/mlflow/client.d.ts +51 -0
- package/dist/connectors/mlflow/client.d.ts.map +1 -0
- package/dist/connectors/mlflow/client.js +93 -0
- package/dist/connectors/mlflow/client.js.map +1 -0
- package/dist/evals/define-eval.d.ts +26 -0
- package/dist/evals/define-eval.d.ts.map +1 -0
- package/dist/evals/define-eval.js +28 -0
- package/dist/evals/define-eval.js.map +1 -0
- package/dist/evals/discover.d.ts +20 -0
- package/dist/evals/discover.d.ts.map +1 -0
- package/dist/evals/discover.js +49 -0
- package/dist/evals/discover.js.map +1 -0
- package/dist/evals/http-driver.d.ts +33 -0
- package/dist/evals/http-driver.d.ts.map +1 -0
- package/dist/evals/http-driver.js +118 -0
- package/dist/evals/http-driver.js.map +1 -0
- package/dist/evals/index.js +13 -0
- package/dist/evals/judge.d.ts +26 -0
- package/dist/evals/judge.d.ts.map +1 -0
- package/dist/evals/judge.js +77 -0
- package/dist/evals/judge.js.map +1 -0
- package/dist/evals/matchers.d.ts +12 -0
- package/dist/evals/matchers.d.ts.map +1 -0
- package/dist/evals/matchers.js +26 -0
- package/dist/evals/matchers.js.map +1 -0
- package/dist/evals/mlflow-report.d.ts +36 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -0
- package/dist/evals/mlflow-report.js +161 -0
- package/dist/evals/mlflow-report.js.map +1 -0
- package/dist/evals/mlflow-run.d.ts +11 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -0
- package/dist/evals/mlflow-run.js +101 -0
- package/dist/evals/mlflow-run.js.map +1 -0
- package/dist/evals/pool.js +24 -0
- package/dist/evals/pool.js.map +1 -0
- package/dist/evals/report.d.ts +25 -0
- package/dist/evals/report.d.ts.map +1 -0
- package/dist/evals/report.js +57 -0
- package/dist/evals/report.js.map +1 -0
- package/dist/evals/run-eval.d.ts +20 -0
- package/dist/evals/run-eval.d.ts.map +1 -0
- package/dist/evals/run-eval.js +144 -0
- package/dist/evals/run-eval.js.map +1 -0
- package/dist/evals/run-evals.d.ts +85 -0
- package/dist/evals/run-evals.d.ts.map +1 -0
- package/dist/evals/run-evals.js +178 -0
- package/dist/evals/run-evals.js.map +1 -0
- package/dist/evals/types.d.ts +127 -0
- package/dist/evals/types.d.ts.map +1 -0
- package/dist/plugins/agents/agents.js +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +33 -33
- package/docs/api/appkit/Class.MlflowClient.md +103 -0
- package/docs/api/appkit/Function.buildAssessments.md +16 -0
- package/docs/api/appkit/Function.configureJudge.md +18 -0
- package/docs/api/appkit/Function.createHttpDriver.md +18 -0
- package/docs/api/appkit/Function.defineEval.md +35 -0
- package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
- package/docs/api/appkit/Function.equals.md +18 -0
- package/docs/api/appkit/Function.evalGlyph.md +18 -0
- package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
- package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
- package/docs/api/appkit/Function.formatEvalResults.md +18 -0
- package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
- package/docs/api/appkit/Function.includes.md +18 -0
- package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
- package/docs/api/appkit/Function.matches.md +18 -0
- package/docs/api/appkit/Function.normalizeHost.md +18 -0
- package/docs/api/appkit/Function.reportToMlflow.md +23 -0
- package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
- package/docs/api/appkit/Function.runEval.md +19 -0
- package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
- package/docs/api/appkit/Function.summarize.md +16 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
- package/docs/api/appkit/Interface.AssertionResult.md +48 -0
- package/docs/api/appkit/Interface.Assessment.md +83 -0
- package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
- package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
- package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
- package/docs/api/appkit/Interface.DriveResult.md +58 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +46 -0
- package/docs/api/appkit/Interface.EvalDriver.md +22 -0
- package/docs/api/appkit/Interface.EvalResult.md +83 -0
- package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
- package/docs/api/appkit/Interface.EvalSummary.md +48 -0
- package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
- package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
- package/docs/api/appkit/Interface.JudgeScore.md +21 -0
- package/docs/api/appkit/Interface.MatchResult.md +34 -0
- package/docs/api/appkit/Interface.PostResult.md +30 -0
- package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
- package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +192 -0
- package/docs/api/appkit/Interface.TestContext.md +208 -0
- package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
- package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
- package/docs/api/appkit/TypeAlias.Severity.md +8 -0
- package/docs/api/appkit.md +63 -17
- package/llms.txt +46 -0
- package/package.json +2 -1
- package/sbom.cdx.json +1 -1
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
import { judgeClosedQA, judgeCustom, judgeFactuality } from "./judge.js";
|
|
2
|
+
|
|
3
|
+
//#region src/evals/run-eval.ts
|
|
4
|
+
/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */
|
|
5
|
+
const DEFAULT_JUDGE_THRESHOLD = .5;
|
|
6
|
+
/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */
|
|
7
|
+
var SkipSignal = class extends Error {
|
|
8
|
+
constructor(reason) {
|
|
9
|
+
super("eval skipped");
|
|
10
|
+
this.reason = reason;
|
|
11
|
+
this.name = "SkipSignal";
|
|
12
|
+
}
|
|
13
|
+
};
|
|
14
|
+
/**
|
|
15
|
+
* Run a single eval against a driver. Never throws for assertion or agent
|
|
16
|
+
* failures — those become a non-passing {@link EvalResult}. Only a malformed
|
|
17
|
+
* eval definition surfaces as `result.error`.
|
|
18
|
+
*/
|
|
19
|
+
async function runEval(def, options) {
|
|
20
|
+
const assertions = [];
|
|
21
|
+
let reply = "";
|
|
22
|
+
let lastInput = "";
|
|
23
|
+
let toolCalls = [];
|
|
24
|
+
let sessionId;
|
|
25
|
+
let lastTraceId;
|
|
26
|
+
let lastSucceeded = false;
|
|
27
|
+
const record = (label, pass, score, detail) => {
|
|
28
|
+
const result = {
|
|
29
|
+
label,
|
|
30
|
+
severity: "gate",
|
|
31
|
+
pass,
|
|
32
|
+
score,
|
|
33
|
+
detail
|
|
34
|
+
};
|
|
35
|
+
assertions.push(result);
|
|
36
|
+
const handle = {
|
|
37
|
+
gate() {
|
|
38
|
+
result.severity = "gate";
|
|
39
|
+
return handle;
|
|
40
|
+
},
|
|
41
|
+
soft() {
|
|
42
|
+
result.severity = "soft";
|
|
43
|
+
return handle;
|
|
44
|
+
},
|
|
45
|
+
atLeast(threshold) {
|
|
46
|
+
result.severity = "soft";
|
|
47
|
+
result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;
|
|
48
|
+
return handle;
|
|
49
|
+
}
|
|
50
|
+
};
|
|
51
|
+
return handle;
|
|
52
|
+
};
|
|
53
|
+
const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale).soft();
|
|
54
|
+
const t = {
|
|
55
|
+
async send(message) {
|
|
56
|
+
lastInput = message;
|
|
57
|
+
const r = await options.driver.send(message);
|
|
58
|
+
reply = r.reply;
|
|
59
|
+
toolCalls = r.toolCalls;
|
|
60
|
+
sessionId = r.sessionId;
|
|
61
|
+
lastSucceeded = r.succeeded;
|
|
62
|
+
if (r.traceId) lastTraceId = r.traceId;
|
|
63
|
+
},
|
|
64
|
+
get reply() {
|
|
65
|
+
return reply;
|
|
66
|
+
},
|
|
67
|
+
get toolCalls() {
|
|
68
|
+
return toolCalls;
|
|
69
|
+
},
|
|
70
|
+
get sessionId() {
|
|
71
|
+
return sessionId;
|
|
72
|
+
},
|
|
73
|
+
succeeded() {
|
|
74
|
+
return record("succeeded", lastSucceeded, void 0, lastSucceeded ? void 0 : "agent turn did not complete successfully");
|
|
75
|
+
},
|
|
76
|
+
calledTool(name) {
|
|
77
|
+
return record(`calledTool(${name})`, toolCalls.includes(name), void 0, `expected tool "${name}" to be called (called: ${toolCalls.length ? toolCalls.join(", ") : "none"})`);
|
|
78
|
+
},
|
|
79
|
+
check(value, matcher) {
|
|
80
|
+
const m = matcher(value);
|
|
81
|
+
return record("check", m.pass, m.score, m.detail);
|
|
82
|
+
},
|
|
83
|
+
judge: {
|
|
84
|
+
async factuality(expected) {
|
|
85
|
+
const { score, rationale } = await judgeFactuality({
|
|
86
|
+
input: lastInput,
|
|
87
|
+
output: reply,
|
|
88
|
+
expected
|
|
89
|
+
});
|
|
90
|
+
return recordJudge("judge.factuality", score, rationale);
|
|
91
|
+
},
|
|
92
|
+
async closedQA(criteria) {
|
|
93
|
+
const { score, rationale } = await judgeClosedQA({
|
|
94
|
+
input: lastInput,
|
|
95
|
+
output: reply,
|
|
96
|
+
criteria
|
|
97
|
+
});
|
|
98
|
+
return recordJudge("judge.closedQA", score, rationale);
|
|
99
|
+
},
|
|
100
|
+
async custom(spec) {
|
|
101
|
+
const { score, rationale } = await judgeCustom(spec, {
|
|
102
|
+
input: lastInput,
|
|
103
|
+
output: reply
|
|
104
|
+
});
|
|
105
|
+
return recordJudge(`judge.${spec.name}`, score, rationale);
|
|
106
|
+
}
|
|
107
|
+
},
|
|
108
|
+
skip(reason) {
|
|
109
|
+
throw new SkipSignal(reason);
|
|
110
|
+
}
|
|
111
|
+
};
|
|
112
|
+
try {
|
|
113
|
+
await def.test(t);
|
|
114
|
+
} catch (err) {
|
|
115
|
+
if (err instanceof SkipSignal) return {
|
|
116
|
+
id: options.id,
|
|
117
|
+
description: def.description,
|
|
118
|
+
skipped: { reason: err.reason },
|
|
119
|
+
assertions,
|
|
120
|
+
passed: true,
|
|
121
|
+
traceId: lastTraceId
|
|
122
|
+
};
|
|
123
|
+
return {
|
|
124
|
+
id: options.id,
|
|
125
|
+
description: def.description,
|
|
126
|
+
assertions,
|
|
127
|
+
passed: false,
|
|
128
|
+
error: err instanceof Error ? err.message : String(err),
|
|
129
|
+
traceId: lastTraceId
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
const passed = assertions.every((a) => a.pass || a.severity === "soft" && !options.strict);
|
|
133
|
+
return {
|
|
134
|
+
id: options.id,
|
|
135
|
+
description: def.description,
|
|
136
|
+
assertions,
|
|
137
|
+
passed,
|
|
138
|
+
traceId: lastTraceId
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
//#endregion
|
|
143
|
+
export { runEval };
|
|
144
|
+
//# sourceMappingURL=run-eval.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n result.severity = \"soft\";\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and soft by default; the caller chains\n // `.atLeast(n)` to set the pass threshold or `.gate()` to promote.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale).soft();\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message);\n reply = r.reply;\n toolCalls = r.toolCalls;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (r.traceId) lastTraceId = r.traceId;\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n try {\n await def.test(t);\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n };\n}\n"],"mappings":";;;;AAYA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAkBhB,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AACzB,WAAO,WAAW;AAClB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAKT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU,CAAC,MAAM;CAE1E,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,QAAQ;AAC5C,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;AAED,KAAI;AACF,QAAM,IAAI,KAAK,EAAE;UACV,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;;CAGH,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EACV"}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import { EvalResult } from "./types.js";
|
|
2
|
+
import { ReportOutcome } from "./mlflow-report.js";
|
|
3
|
+
import { FinishOutcome } from "./mlflow-run.js";
|
|
4
|
+
|
|
5
|
+
//#region src/evals/run-evals.d.ts
|
|
6
|
+
interface RunEvalsOptions {
|
|
7
|
+
/** Project root containing `server/agents/`. Defaults to `process.cwd()`. */
|
|
8
|
+
rootDir?: string;
|
|
9
|
+
/** Base URL of the running app to drive, e.g. `http://localhost:3000`. */
|
|
10
|
+
baseUrl: string;
|
|
11
|
+
/** Substring filter on `<agent>/<id>` (or an exact agent id). */
|
|
12
|
+
filter?: string;
|
|
13
|
+
/** Soft assertion failures also fail the eval. */
|
|
14
|
+
strict?: boolean;
|
|
15
|
+
/** Extra request headers for the driver (e.g. auth for a deployed app). */
|
|
16
|
+
headers?: Record<string, string>;
|
|
17
|
+
/** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */
|
|
18
|
+
timeoutMs?: number;
|
|
19
|
+
/**
|
|
20
|
+
* Max evals to drive concurrently. Each eval opens one stream to the app as
|
|
21
|
+
* the same user, so keep this at or below the app's
|
|
22
|
+
* `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the
|
|
23
|
+
* 429 guard. Defaults to 4; clamped to `[1, total]`.
|
|
24
|
+
*/
|
|
25
|
+
concurrency?: number;
|
|
26
|
+
/**
|
|
27
|
+
* When set, create a native MLflow "Evaluation run": each eval's trace is
|
|
28
|
+
* linked to the run, pass/fail is written as feedback, and aggregate metrics
|
|
29
|
+
* are logged. Requires Databricks creds + the target experiment.
|
|
30
|
+
*/
|
|
31
|
+
mlflow?: {
|
|
32
|
+
host: string;
|
|
33
|
+
token: string;
|
|
34
|
+
experimentId: string; /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */
|
|
35
|
+
sqlWarehouseId?: string;
|
|
36
|
+
};
|
|
37
|
+
/**
|
|
38
|
+
* When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a
|
|
39
|
+
* Databricks serving endpoint (`model`).
|
|
40
|
+
*/
|
|
41
|
+
judge?: {
|
|
42
|
+
host: string;
|
|
43
|
+
token: string;
|
|
44
|
+
model: string;
|
|
45
|
+
};
|
|
46
|
+
/** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */
|
|
47
|
+
now?: number;
|
|
48
|
+
/** Progress callback, invoked as evals are discovered, started, and finished. */
|
|
49
|
+
onEvent?: (event: EvalProgress) => void;
|
|
50
|
+
}
|
|
51
|
+
type EvalProgress = {
|
|
52
|
+
type: "discovered";
|
|
53
|
+
total: number;
|
|
54
|
+
} | {
|
|
55
|
+
type: "run-created";
|
|
56
|
+
runId: string;
|
|
57
|
+
} | {
|
|
58
|
+
type: "start";
|
|
59
|
+
id: string;
|
|
60
|
+
index: number;
|
|
61
|
+
total: number;
|
|
62
|
+
} | {
|
|
63
|
+
type: "result";
|
|
64
|
+
result: EvalResult;
|
|
65
|
+
index: number;
|
|
66
|
+
total: number;
|
|
67
|
+
};
|
|
68
|
+
interface EvalRunSummary {
|
|
69
|
+
results: EvalResult[];
|
|
70
|
+
/** Present when an MLflow evaluation run was created. */
|
|
71
|
+
mlflow?: {
|
|
72
|
+
runId: string;
|
|
73
|
+
report: ReportOutcome;
|
|
74
|
+
finish: FinishOutcome;
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
79
|
+
* the agents on a running app. Never throws for an individual eval — load/run
|
|
80
|
+
* failures become non-passing {@link EvalResult}s.
|
|
81
|
+
*/
|
|
82
|
+
declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
|
|
83
|
+
//#endregion
|
|
84
|
+
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir };
|
|
85
|
+
//# sourceMappingURL=run-evals.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;UAYiB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAoC8B;EAlC9B,MAAA;EAFA;EAIA,MAAA;EAAA;EAEA,OAAA,GAAU,MAAA;EAAA;EAEV,SAAA;EAOA;;;;;;EAAA,WAAA;EAiBU;;;;;EAXV,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UAeQ;IAbR,cAAA;EAAA;EAiBoC;;;;EAXtC,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EAWnC;EATJ,GAAA;EAS4B;EAP5B,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EAE6D;EAAtE,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;iBAqIrC,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
|
+
import { discoverEvalFiles } from "./discover.js";
|
|
3
|
+
import { createHttpDriver } from "./http-driver.js";
|
|
4
|
+
import { configureJudge, teardownJudge } from "./judge.js";
|
|
5
|
+
import { mapPool } from "./pool.js";
|
|
6
|
+
import { reportToMlflow } from "./mlflow-report.js";
|
|
7
|
+
import { runEval } from "./run-eval.js";
|
|
8
|
+
import { createEvalRun, finishEvalRun } from "./mlflow-run.js";
|
|
9
|
+
import { pathToFileURL } from "node:url";
|
|
10
|
+
|
|
11
|
+
//#region src/evals/run-evals.ts
|
|
12
|
+
/**
|
|
13
|
+
* Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.
|
|
14
|
+
* Uses tsx's programmatic loader so TypeScript eval files run without a build
|
|
15
|
+
* step. The specifier is indirected so the type checker doesn't try to resolve
|
|
16
|
+
* tsx's internal entry.
|
|
17
|
+
*/
|
|
18
|
+
async function loadEval(file) {
|
|
19
|
+
const tsxApi = "tsx/esm/api";
|
|
20
|
+
let tsImport;
|
|
21
|
+
try {
|
|
22
|
+
({tsImport} = await import(tsxApi));
|
|
23
|
+
} catch {
|
|
24
|
+
throw new Error("Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).");
|
|
25
|
+
}
|
|
26
|
+
const def = resolveEvalDefault(await tsImport(pathToFileURL(file).href, import.meta.url));
|
|
27
|
+
if (!def) throw new Error(`${file}: must default-export defineEval({ test })`);
|
|
28
|
+
return def;
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* Unwrap the eval default export across module-interop shapes. Depending on
|
|
32
|
+
* whether the eval file is treated as ESM or CJS, the value lands at
|
|
33
|
+
* `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or
|
|
34
|
+
* `mod` itself. Returns the first candidate that looks like an eval.
|
|
35
|
+
*/
|
|
36
|
+
function resolveEvalDefault(mod) {
|
|
37
|
+
let candidate = mod;
|
|
38
|
+
for (let i = 0; i < 4 && candidate; i++) {
|
|
39
|
+
if (typeof candidate.test === "function") return candidate;
|
|
40
|
+
candidate = candidate.default;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Load and run a single discovered eval. Never throws — a load/run failure
|
|
45
|
+
* becomes a non-passing {@link EvalResult} so one bad eval can't abort the run.
|
|
46
|
+
*/
|
|
47
|
+
async function runOne(d, id, runId, options) {
|
|
48
|
+
try {
|
|
49
|
+
const def = await loadEval(d.file);
|
|
50
|
+
return await runEval(def, {
|
|
51
|
+
id,
|
|
52
|
+
driver: createHttpDriver({
|
|
53
|
+
baseUrl: options.baseUrl,
|
|
54
|
+
agent: def.agent ?? d.agent,
|
|
55
|
+
headers: options.headers,
|
|
56
|
+
mlflowRunId: runId,
|
|
57
|
+
timeoutMs: options.timeoutMs
|
|
58
|
+
}),
|
|
59
|
+
strict: options.strict
|
|
60
|
+
});
|
|
61
|
+
} catch (err) {
|
|
62
|
+
return {
|
|
63
|
+
id,
|
|
64
|
+
assertions: [],
|
|
65
|
+
passed: false,
|
|
66
|
+
error: err instanceof Error ? err.message : String(err)
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */
|
|
71
|
+
async function maybeConfigureJudge(options) {
|
|
72
|
+
if (!options.judge) return;
|
|
73
|
+
await configureJudge({
|
|
74
|
+
client: new MlflowClient(options.judge.host, options.judge.token),
|
|
75
|
+
token: options.judge.token,
|
|
76
|
+
model: options.judge.model
|
|
77
|
+
});
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Report per-eval assessments and finish the MLflow run, when one was created.
|
|
81
|
+
* Returns the run summary, or `undefined` when there was no run to finalize.
|
|
82
|
+
*/
|
|
83
|
+
async function finalizeMlflow(client, runId, results, options) {
|
|
84
|
+
if (!client || !runId) return void 0;
|
|
85
|
+
let report = {
|
|
86
|
+
written: 0,
|
|
87
|
+
skipped: 0,
|
|
88
|
+
failures: []
|
|
89
|
+
};
|
|
90
|
+
try {
|
|
91
|
+
report = await reportToMlflow(client, results, options.mlflow?.sqlWarehouseId);
|
|
92
|
+
} catch (err) {
|
|
93
|
+
report.failures.push({
|
|
94
|
+
traceId: "(report)",
|
|
95
|
+
error: err instanceof Error ? err.message : String(err)
|
|
96
|
+
});
|
|
97
|
+
}
|
|
98
|
+
const finish = await finishEvalRun(client, {
|
|
99
|
+
runId,
|
|
100
|
+
results,
|
|
101
|
+
endTime: options.now ?? Date.now()
|
|
102
|
+
});
|
|
103
|
+
return {
|
|
104
|
+
runId,
|
|
105
|
+
report,
|
|
106
|
+
finish
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Default max evals in flight. Each eval opens one stream to the app as the
|
|
111
|
+
* same user; the server caps concurrent streams per user at 5 by default
|
|
112
|
+
* (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.
|
|
113
|
+
*/
|
|
114
|
+
const DEFAULT_CONCURRENCY = 4;
|
|
115
|
+
/**
|
|
116
|
+
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
117
|
+
* the agents on a running app. Never throws for an individual eval — load/run
|
|
118
|
+
* failures become non-passing {@link EvalResult}s.
|
|
119
|
+
*/
|
|
120
|
+
async function runEvalsInDir(options) {
|
|
121
|
+
const root = options.rootDir ?? process.cwd();
|
|
122
|
+
const now = options.now ?? Date.now();
|
|
123
|
+
let discovered = discoverEvalFiles(root);
|
|
124
|
+
if (options.filter) {
|
|
125
|
+
const f = options.filter;
|
|
126
|
+
discovered = discovered.filter((d) => d.agent === f || `${d.agent}/${d.id}`.includes(f));
|
|
127
|
+
}
|
|
128
|
+
const emit = options.onEvent ?? (() => {});
|
|
129
|
+
const total = discovered.length;
|
|
130
|
+
emit({
|
|
131
|
+
type: "discovered",
|
|
132
|
+
total
|
|
133
|
+
});
|
|
134
|
+
await maybeConfigureJudge(options);
|
|
135
|
+
try {
|
|
136
|
+
let runId;
|
|
137
|
+
let mlflowClient;
|
|
138
|
+
if (options.mlflow) {
|
|
139
|
+
mlflowClient = new MlflowClient(options.mlflow.host, options.mlflow.token);
|
|
140
|
+
runId = await createEvalRun(mlflowClient, {
|
|
141
|
+
experimentId: options.mlflow.experimentId,
|
|
142
|
+
runName: `appkit-eval ${new Date(now).toISOString()}`,
|
|
143
|
+
startTime: now
|
|
144
|
+
});
|
|
145
|
+
emit({
|
|
146
|
+
type: "run-created",
|
|
147
|
+
runId
|
|
148
|
+
});
|
|
149
|
+
}
|
|
150
|
+
const results = await mapPool(discovered, options.concurrency ?? DEFAULT_CONCURRENCY, async (d, index) => {
|
|
151
|
+
const id = `${d.agent}/${d.id}`;
|
|
152
|
+
emit({
|
|
153
|
+
type: "start",
|
|
154
|
+
id,
|
|
155
|
+
index,
|
|
156
|
+
total
|
|
157
|
+
});
|
|
158
|
+
const result = await runOne(d, id, runId, options);
|
|
159
|
+
emit({
|
|
160
|
+
type: "result",
|
|
161
|
+
result,
|
|
162
|
+
index,
|
|
163
|
+
total
|
|
164
|
+
});
|
|
165
|
+
return result;
|
|
166
|
+
});
|
|
167
|
+
const summary = { results };
|
|
168
|
+
const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);
|
|
169
|
+
if (mlflow) summary.mlflow = mlflow;
|
|
170
|
+
return summary;
|
|
171
|
+
} finally {
|
|
172
|
+
teardownJudge();
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
//#endregion
|
|
177
|
+
export { runEvalsInDir };
|
|
178
|
+
//# sourceMappingURL=run-evals.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-evals.js","names":[],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport { type DiscoveredEval, discoverEvalFiles } from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */\n timeoutMs?: number;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n * Uses tsx's programmatic loader so TypeScript eval files run without a build\n * step. The specifier is indirected so the type checker doesn't try to resolve\n * tsx's internal entry.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n\n const mod = await tsImport(pathToFileURL(file).href, import.meta.url);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Load and run a single discovered eval. Never throws — a load/run failure\n * becomes a non-passing {@link EvalResult} so one bad eval can't abort the run.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n const def = await loadEval(d.file);\n const driver = createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n timeoutMs: options.timeoutMs,\n });\n return await runEval(def, { id, driver, strict: options.strict });\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n const total = discovered.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run evals through a bounded pool so independent turns overlap instead of\n // summing their latencies. runOne never throws, so a pool worker never\n // rejects; results preserve discovery order (mapPool writes by index).\n const results = await mapPool(\n discovered,\n options.concurrency ?? DEFAULT_CONCURRENCY,\n async (d, index) => {\n const id = `${d.agent}/${d.id}`;\n emit({ type: \"start\", id, index, total });\n const result = await runOne(d, id, runId, options);\n emit({ type: \"result\", result, index, total });\n return result;\n },\n );\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;AAyEA,eAAe,SAAS,MAAuC;CAC7D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;CAIH,MAAM,MAAM,mBADA,MAAM,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI,CAClC;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;;AAST,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;AASrD,eAAe,OACb,GACA,IACA,OACA,SACqB;AACrB,KAAI;EACF,MAAM,MAAM,MAAM,SAAS,EAAE,KAAK;AAQlC,SAAO,MAAM,QAAQ,KAAK;GAAE;GAAI,QAPjB,iBAAiB;IAC9B,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IACb,WAAW,QAAQ;IACpB,CAAC;GACsC,QAAQ,QAAQ;GAAQ,CAAC;UAC1D,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;AAKL,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;AAO5B,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CACvC,MAAM,QAAQ,WAAW;AACzB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAMtC,MAAM,UAAU,MAAM,QACpB,YACA,QAAQ,eAAe,qBACvB,OAAO,GAAG,UAAU;GAClB,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;AAC3B,QAAK;IAAE,MAAM;IAAS;IAAI;IAAO;IAAO,CAAC;GACzC,MAAM,SAAS,MAAM,OAAO,GAAG,IAAI,OAAO,QAAQ;AAClD,QAAK;IAAE,MAAM;IAAU;IAAQ;IAAO;IAAO,CAAC;AAC9C,UAAO;IAEV;EAED,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
//#region src/evals/types.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* Agent evaluation primitives — an eve-style authoring API that runs against
|
|
4
|
+
* AppKit agents and reports to Databricks MLflow.
|
|
5
|
+
*
|
|
6
|
+
* Evals live in `server/agents/<id>/evals/*.eval.ts`, each default-exporting a
|
|
7
|
+
* {@link EvalDefinition} via {@link defineEval}. A runner drives the agent
|
|
8
|
+
* (today over HTTP against a running app), and the `test` function asserts on
|
|
9
|
+
* the reply and tool usage with deterministic matchers (and, later, LLM judges).
|
|
10
|
+
*/
|
|
11
|
+
/** Result of a deterministic matcher run against a value. */
|
|
12
|
+
interface MatchResult {
|
|
13
|
+
pass: boolean;
|
|
14
|
+
/** Optional 0..1 score for scored matchers (similarity, judges). */
|
|
15
|
+
score?: number;
|
|
16
|
+
/** Human-readable explanation, shown on failure. */
|
|
17
|
+
detail?: string;
|
|
18
|
+
}
|
|
19
|
+
/** A deterministic matcher: inspects a string value and returns a result. */
|
|
20
|
+
type Matcher = (value: string) => MatchResult;
|
|
21
|
+
/** Whether an assertion fails the eval (`gate`) or is tracked only (`soft`). */
|
|
22
|
+
type Severity = "gate" | "soft";
|
|
23
|
+
/** A single recorded assertion outcome. */
|
|
24
|
+
interface AssertionResult {
|
|
25
|
+
label: string;
|
|
26
|
+
severity: Severity;
|
|
27
|
+
pass: boolean;
|
|
28
|
+
score?: number;
|
|
29
|
+
detail?: string;
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Chainable handle returned by every assertion to control its severity.
|
|
33
|
+
* Mirrors eve: assertions are gates by default; `.soft()` demotes to a tracked
|
|
34
|
+
* metric; `.atLeast(n)` is a soft, score-thresholded assertion.
|
|
35
|
+
*/
|
|
36
|
+
interface AssertionHandle {
|
|
37
|
+
/** Promote to a hard gate — failure fails the eval (non-zero exit). */
|
|
38
|
+
gate(): AssertionHandle;
|
|
39
|
+
/** Demote to a tracked metric — doesn't fail unless running with `strict`. */
|
|
40
|
+
soft(): AssertionHandle;
|
|
41
|
+
/** Soft assertion that passes only when the score is at least `threshold`. */
|
|
42
|
+
atLeast(threshold: number): AssertionHandle;
|
|
43
|
+
}
|
|
44
|
+
/** What a driver returns for a single `t.send`. */
|
|
45
|
+
interface DriveResult {
|
|
46
|
+
/** The final assistant message text. */
|
|
47
|
+
reply: string;
|
|
48
|
+
/** Names of tools the agent called during the turn. */
|
|
49
|
+
toolCalls: string[];
|
|
50
|
+
/** Whether the turn completed without an agent/stream error. */
|
|
51
|
+
succeeded: boolean;
|
|
52
|
+
/** Thread/session id, when the driver exposes one. */
|
|
53
|
+
sessionId?: string;
|
|
54
|
+
/** MLflow trace id for the turn, when tracing is enabled on the app. */
|
|
55
|
+
traceId?: string;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Abstraction over how the agent is driven. The HTTP driver posts to a running
|
|
59
|
+
* app's agents endpoint; future drivers (in-process) implement the same shape.
|
|
60
|
+
*/
|
|
61
|
+
interface EvalDriver {
|
|
62
|
+
send(message: string): Promise<DriveResult>;
|
|
63
|
+
}
|
|
64
|
+
/** The `t` context passed to an eval's `test` function. */
|
|
65
|
+
interface TestContext {
|
|
66
|
+
/** Send a user message to the agent and capture its response. */
|
|
67
|
+
send(message: string): Promise<void>;
|
|
68
|
+
/** The last assistant reply. */
|
|
69
|
+
readonly reply: string;
|
|
70
|
+
/** Tools called during the last turn. */
|
|
71
|
+
readonly toolCalls: string[];
|
|
72
|
+
/** The current session/thread id, if any. */
|
|
73
|
+
readonly sessionId: string | undefined;
|
|
74
|
+
/** Assert the last turn completed successfully (gate by default). */
|
|
75
|
+
succeeded(): AssertionHandle;
|
|
76
|
+
/** Assert a tool was called during the run (gate by default). */
|
|
77
|
+
calledTool(name: string): AssertionHandle;
|
|
78
|
+
/** Assert a value against a matcher, e.g. `t.check(t.reply, includes("Sunny"))`. */
|
|
79
|
+
check(value: string, matcher: Matcher): AssertionHandle;
|
|
80
|
+
/**
|
|
81
|
+
* LLM-as-judge scoring of the last reply (via autoevals → a Databricks judge
|
|
82
|
+
* model). Each returns a scored, soft-by-default assertion; chain `.atLeast(n)`
|
|
83
|
+
* to set the pass threshold or `.gate()` to make it a hard gate. Requires the
|
|
84
|
+
* judge to be configured (`--judge-model`).
|
|
85
|
+
*/
|
|
86
|
+
judge: {
|
|
87
|
+
/** Score factuality of the reply against an expected reference. */factuality(expected: string): Promise<AssertionHandle>; /** Score whether the reply answers the question, per optional `criteria`. */
|
|
88
|
+
closedQA(criteria: string): Promise<AssertionHandle>; /** A custom prompt-template judge (the TS analog of MLflow's `@scorer`). */
|
|
89
|
+
custom(spec: CustomJudgeSpec): Promise<AssertionHandle>;
|
|
90
|
+
};
|
|
91
|
+
/** Skip this eval with an optional reason. */
|
|
92
|
+
skip(reason?: string): never;
|
|
93
|
+
}
|
|
94
|
+
/** A custom LLM-judge definition: a prompt template and choice→score mapping. */
|
|
95
|
+
interface CustomJudgeSpec {
|
|
96
|
+
name: string;
|
|
97
|
+
promptTemplate: string;
|
|
98
|
+
choiceScores: Record<string, number>;
|
|
99
|
+
}
|
|
100
|
+
/** A single eval, default-exported from a `*.eval.ts` file. */
|
|
101
|
+
interface EvalDefinition {
|
|
102
|
+
/** Short human description, shown in reports. */
|
|
103
|
+
description?: string;
|
|
104
|
+
/** Target agent id. Defaults to the eval's parent `server/agents/<id>` dir. */
|
|
105
|
+
agent?: string;
|
|
106
|
+
/** The eval body: drive the agent and assert on its behavior. */
|
|
107
|
+
test(t: TestContext): Promise<void> | void;
|
|
108
|
+
}
|
|
109
|
+
/** The outcome of running one eval. */
|
|
110
|
+
interface EvalResult {
|
|
111
|
+
id: string;
|
|
112
|
+
description?: string;
|
|
113
|
+
/** Set when the eval called `t.skip`. */
|
|
114
|
+
skipped?: {
|
|
115
|
+
reason?: string;
|
|
116
|
+
};
|
|
117
|
+
assertions: AssertionResult[];
|
|
118
|
+
/** True when all gates passed (and, under strict, all soft assertions too). */
|
|
119
|
+
passed: boolean;
|
|
120
|
+
/** Set when the eval threw before completing. */
|
|
121
|
+
error?: string;
|
|
122
|
+
/** MLflow trace id of the eval's last turn, for attaching assessments. */
|
|
123
|
+
traceId?: string;
|
|
124
|
+
}
|
|
125
|
+
//#endregion
|
|
126
|
+
export { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext };
|
|
127
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;EAIoB;EAF5B,IAAA,IAAQ,eAAA;EAEmC;EAA3C,OAAA,CAAQ,SAAA,WAAoB,eAAA;AAAA;;UAIb,WAAA;EAJf;EAMA,KAAA;EAN4B;EAQ5B,SAAA;EAR2C;EAU3C,SAAA;EAN0B;EAQ1B,SAAA;EAR0B;EAU1B,OAAA;AAAA;;;;;UAOe,UAAA;EACf,IAAA,CAAK,OAAA,WAAkB,OAAA,CAAQ,WAAA;AAAA;;UAIhB,WAAA;EAJf;EAMA,IAAA,CAAK,OAAA,WAAkB,OAAA;EANA;EAAA,SAQd,KAAA;EARiC;EAAA,SAUjC,SAAA;EANM;EAAA,SAQN,SAAA;;EAET,SAAA,IAAa,eAAA;EAAA;EAEb,UAAA,CAAW,IAAA,WAAe,eAAA;EAEI;EAA9B,KAAA,CAAM,KAAA,UAAe,OAAA,EAAS,OAAA,GAAU,eAAA;EASA;;;;;;EAFxC,KAAA;IAMwC,mEAJtC,UAAA,CAAW,QAAA,WAAmB,OAAA,CAAQ,eAAA,GArBxC;IAuBE,QAAA,CAAS,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAvBf;IAyBrB,MAAA,CAAO,IAAA,EAAM,eAAA,GAAkB,OAAA,CAAQ,eAAA;EAAA;EAnBhC;EAsBT,IAAA,CAAK,MAAA;AAAA;;UAIU,eAAA;EACf,IAAA;EACA,cAAA;EACA,YAAA,EAAc,MAAA;AAAA;;UAIC,cAAA;EApBf;EAsBA,WAAA;EApBa;EAsBb,KAAA;EAtBwC;EAwBxC,IAAA,CAAK,CAAA,EAAG,WAAA,GAAc,OAAA;AAAA;;UAIP,UAAA;EACf,EAAA;EACA,WAAA;EA1BS;EA4BT,OAAA;IAAY,MAAA;EAAA;EACZ,UAAA,EAAY,eAAA;EA1BQ;EA4BpB,MAAA;EAxBe;EA0Bf,KAAA;;EAEA,OAAA;AAAA"}
|
|
@@ -17,8 +17,8 @@ import { functionToolToDefinition, isFunctionTool } from "../../core/agent/tools
|
|
|
17
17
|
import { isHostedTool, resolveHostedTools } from "../../core/agent/tools/hosted-tools.js";
|
|
18
18
|
import { isToolkitEntry } from "../../core/agent/types.js";
|
|
19
19
|
import "../../core/agent/tools/index.js";
|
|
20
|
-
import { loadAgentsFromDir } from "../../core/agent/load-agents.js";
|
|
21
20
|
import { CODE_AGENTS_SOURCE_DIR } from "../../core/agent/load-code-agents.js";
|
|
21
|
+
import { loadAgentsFromDir } from "../../core/agent/load-agents.js";
|
|
22
22
|
import { ActiveStreamTracker } from "./active-stream-tracker.js";
|
|
23
23
|
import { buildAdapterExtensions, supervisorToolDescription, warnOnCapabilityMismatch } from "./adapter-extensions.js";
|
|
24
24
|
import { requiresApproval } from "./approval.js";
|