@databricks/appkit 0.73.0 → 0.74.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +7 -0
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +6 -6
- package/dist/beta.js +6 -6
- package/dist/cli/commands/agent/eval.js +85 -21
- package/dist/cli/commands/agent/eval.js.map +1 -1
- package/dist/cli/commands/registry/add.js +77 -6
- package/dist/cli/commands/registry/add.js.map +1 -1
- package/dist/evals/dataset.d.ts +14 -1
- package/dist/evals/dataset.d.ts.map +1 -1
- package/dist/evals/dataset.js +16 -1
- package/dist/evals/dataset.js.map +1 -1
- package/dist/evals/define-eval.d.ts +4 -2
- package/dist/evals/define-eval.d.ts.map +1 -1
- package/dist/evals/define-eval.js +5 -1
- package/dist/evals/define-eval.js.map +1 -1
- package/dist/evals/discover.d.ts +15 -1
- package/dist/evals/discover.d.ts.map +1 -1
- package/dist/evals/discover.js +40 -10
- package/dist/evals/discover.js.map +1 -1
- package/dist/evals/http-driver.d.ts.map +1 -1
- package/dist/evals/http-driver.js +31 -9
- package/dist/evals/http-driver.js.map +1 -1
- package/dist/evals/index.d.ts +5 -5
- package/dist/evals/index.js +5 -5
- package/dist/evals/report.d.ts +16 -1
- package/dist/evals/report.d.ts.map +1 -1
- package/dist/evals/report.js +64 -2
- package/dist/evals/report.js.map +1 -1
- package/dist/evals/run-eval.d.ts +5 -0
- package/dist/evals/run-eval.d.ts.map +1 -1
- package/dist/evals/run-eval.js +44 -3
- package/dist/evals/run-eval.js.map +1 -1
- package/dist/evals/run-evals.d.ts +32 -3
- package/dist/evals/run-evals.d.ts.map +1 -1
- package/dist/evals/run-evals.js +128 -26
- package/dist/evals/run-evals.js.map +1 -1
- package/dist/evals/types.d.ts +40 -2
- package/dist/evals/types.d.ts.map +1 -1
- package/dist/registry/manifest-loader.d.ts +1 -1
- package/docs/api/appkit/Function.defineEvalConfig.md +18 -0
- package/docs/api/appkit/Function.discoverEvalConfigs.md +18 -0
- package/docs/api/appkit/Function.formatResultsJUnit.md +18 -0
- package/docs/api/appkit/Function.formatResultsJson.md +18 -0
- package/docs/api/appkit/Function.runWithRetries.md +28 -0
- package/docs/api/appkit/Function.userTurns.md +20 -0
- package/docs/api/appkit/Interface.DiscoveredEvalConfig.md +25 -0
- package/docs/api/appkit/Interface.DriveResult.md +28 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +22 -0
- package/docs/api/appkit/Interface.EvalDriver.md +10 -4
- package/docs/api/appkit/Interface.EvalResult.md +11 -0
- package/docs/api/appkit/Interface.EvalSummary.md +11 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +11 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +23 -1
- package/docs/api/appkit/Interface.TestContext.md +30 -8
- package/docs/api/appkit.md +7 -0
- package/docs/plugins/agents.md +255 -4
- package/llms.txt +7 -0
- package/package.json +1 -1
- package/sbom.cdx.json +1 -1
package/dist/evals/index.d.ts
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth, resolveWorkspaceClient } from "../connectors/mlflow/auth.js";
|
|
2
2
|
import { MlflowClient, PostResult, normalizeHost } from "../connectors/mlflow/client.js";
|
|
3
3
|
import "../connectors/mlflow/index.js";
|
|
4
|
-
import { DatasetRow, ReadEvalDatasetOptions, readEvalDataset } from "./dataset.js";
|
|
4
|
+
import { DatasetRow, ReadEvalDatasetOptions, readEvalDataset, userTurns } from "./dataset.js";
|
|
5
5
|
import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext } from "./types.js";
|
|
6
|
-
import { defineEval } from "./define-eval.js";
|
|
7
|
-
import { DiscoveredEval, discoverEvalFiles } from "./discover.js";
|
|
6
|
+
import { defineEval, defineEvalConfig } from "./define-eval.js";
|
|
7
|
+
import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
|
|
8
8
|
import { HttpDriverOptions, createHttpDriver } from "./http-driver.js";
|
|
9
9
|
import { JudgeConfig, JudgeScore, configureJudge, isJudgeConfigured } from "./judge.js";
|
|
10
10
|
import { equals, includes, matches } from "./matchers.js";
|
|
11
11
|
import { Assessment, ReportOutcome, buildAssessments, reportToMlflow } from "./mlflow-report.js";
|
|
12
|
-
import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize } from "./report.js";
|
|
12
|
+
import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./report.js";
|
|
13
13
|
import { RunEvalOptions, runEval } from "./run-eval.js";
|
|
14
|
-
import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir } from "./run-evals.js";
|
|
14
|
+
import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries } from "./run-evals.js";
|
package/dist/evals/index.js
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { resolveDatabricksAuth, resolveWorkspaceClient } from "../connectors/mlflow/auth.js";
|
|
2
2
|
import { MlflowClient, normalizeHost } from "../connectors/mlflow/client.js";
|
|
3
|
-
import { readEvalDataset } from "./dataset.js";
|
|
4
|
-
import { defineEval } from "./define-eval.js";
|
|
5
|
-
import { discoverEvalFiles } from "./discover.js";
|
|
3
|
+
import { readEvalDataset, userTurns } from "./dataset.js";
|
|
4
|
+
import { defineEval, defineEvalConfig } from "./define-eval.js";
|
|
5
|
+
import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
|
|
6
6
|
import { createHttpDriver } from "./http-driver.js";
|
|
7
7
|
import { configureJudge, isJudgeConfigured } from "./judge.js";
|
|
8
8
|
import { equals, includes, matches } from "./matchers.js";
|
|
9
9
|
import { buildAssessments, reportToMlflow } from "./mlflow-report.js";
|
|
10
|
-
import { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize } from "./report.js";
|
|
10
|
+
import { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./report.js";
|
|
11
11
|
import { runEval } from "./run-eval.js";
|
|
12
|
-
import { runEvalsInDir } from "./run-evals.js";
|
|
12
|
+
import { runEvalsInDir, runWithRetries } from "./run-evals.js";
|
|
13
13
|
|
|
14
14
|
export { };
|
package/dist/evals/report.d.ts
CHANGED
|
@@ -8,6 +8,8 @@ interface EvalSummary {
|
|
|
8
8
|
skipped: number;
|
|
9
9
|
/** True when no eval failed (skips don't count as failures). */
|
|
10
10
|
allPassed: boolean;
|
|
11
|
+
/** Fraction of scored (non-skipped) evals that passed, 0..1 (1 when none scored). */
|
|
12
|
+
passRate: number;
|
|
11
13
|
}
|
|
12
14
|
declare function summarize(results: EvalResult[]): EvalSummary;
|
|
13
15
|
/** Status glyph for a single eval result. */
|
|
@@ -20,6 +22,19 @@ declare function formatEvalDetail(result: EvalResult): string[];
|
|
|
20
22
|
declare function formatSummaryLine(results: EvalResult[]): string;
|
|
21
23
|
/** Render all results as a human-readable console report (non-streaming). */
|
|
22
24
|
declare function formatEvalResults(results: EvalResult[]): string;
|
|
25
|
+
/**
|
|
26
|
+
* Render results as a machine-readable JSON report (2-space indented):
|
|
27
|
+
* `{ summary: EvalSummary, results: EvalResult[] }`. Faithful to the types —
|
|
28
|
+
* every field present on a result round-trips.
|
|
29
|
+
*/
|
|
30
|
+
declare function formatResultsJson(results: EvalResult[]): string;
|
|
31
|
+
/**
|
|
32
|
+
* Render results as JUnit XML for standard CI test reporters: a single
|
|
33
|
+
* `<testsuite name="appkit-agent-evals">` with one `<testcase>` per result.
|
|
34
|
+
* Failures carry a `<failure>` (error or failing-gate summary); skips a
|
|
35
|
+
* `<skipped>`. All attribute/text values are XML-escaped.
|
|
36
|
+
*/
|
|
37
|
+
declare function formatResultsJUnit(results: EvalResult[]): string;
|
|
23
38
|
//#endregion
|
|
24
|
-
export { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize };
|
|
39
|
+
export { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize };
|
|
25
40
|
//# sourceMappingURL=report.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"report.d.ts","names":[],"sources":["../../src/evals/report.ts"],"mappings":";;;UAEiB,WAAA;EACf,KAAA;EACA,MAAA;EACA,MAAA;EACA,OAAA;EAJ0B;EAM1B,SAAA;AAAA;AAAA,iBAGc,SAAA,CAAU,OAAA,EAAS,UAAA,KAAe,WAAA;;
|
|
1
|
+
{"version":3,"file":"report.d.ts","names":[],"sources":["../../src/evals/report.ts"],"mappings":";;;UAEiB,WAAA;EACf,KAAA;EACA,MAAA;EACA,MAAA;EACA,OAAA;EAJ0B;EAM1B,SAAA;EAJA;EAMA,QAAA;AAAA;AAAA,iBAGc,SAAA,CAAU,OAAA,EAAS,UAAA,KAAe,WAAA;;iBAqBlC,SAAA,CAAU,MAAA,EAAQ,UAAA;;iBAMlB,kBAAA,CAAmB,MAAA,EAAQ,UAAA;;iBAY3B,gBAAA,CAAiB,MAAA,EAAQ,UAAA;;iBAYzB,iBAAA,CAAkB,OAAA,EAAS,UAAA;;iBAM3B,iBAAA,CAAkB,OAAA,EAAS,UAAA;;;AApC3C;;;iBAoDgB,iBAAA,CAAkB,OAAA,EAAS,UAAA;;AA9C3C;;;;;iBA2FgB,kBAAA,CAAmB,OAAA,EAAS,UAAA"}
|
package/dist/evals/report.js
CHANGED
|
@@ -6,12 +6,14 @@ function summarize(results) {
|
|
|
6
6
|
for (const r of results) if (r.skipped) skipped++;
|
|
7
7
|
else if (r.passed) passed++;
|
|
8
8
|
else failed++;
|
|
9
|
+
const scored = passed + failed;
|
|
9
10
|
return {
|
|
10
11
|
total: results.length,
|
|
11
12
|
passed,
|
|
12
13
|
failed,
|
|
13
14
|
skipped,
|
|
14
|
-
allPassed: failed === 0
|
|
15
|
+
allPassed: failed === 0,
|
|
16
|
+
passRate: scored === 0 ? 1 : passed / scored
|
|
15
17
|
};
|
|
16
18
|
}
|
|
17
19
|
/** Status glyph for a single eval result. */
|
|
@@ -51,7 +53,67 @@ function formatEvalResults(results) {
|
|
|
51
53
|
lines.push(formatSummaryLine(results));
|
|
52
54
|
return lines.join("\n");
|
|
53
55
|
}
|
|
56
|
+
/**
|
|
57
|
+
* Render results as a machine-readable JSON report (2-space indented):
|
|
58
|
+
* `{ summary: EvalSummary, results: EvalResult[] }`. Faithful to the types —
|
|
59
|
+
* every field present on a result round-trips.
|
|
60
|
+
*/
|
|
61
|
+
function formatResultsJson(results) {
|
|
62
|
+
return JSON.stringify({
|
|
63
|
+
summary: summarize(results),
|
|
64
|
+
results
|
|
65
|
+
}, null, 2);
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Drop the C0 control chars XML 1.0 forbids even when escaped (except tab, LF,
|
|
69
|
+
* CR) — a raw NUL or ANSI escape would otherwise make the JUnit doc reject on parse.
|
|
70
|
+
*/
|
|
71
|
+
function stripXmlControlChars(value) {
|
|
72
|
+
let out = "";
|
|
73
|
+
for (const ch of value) {
|
|
74
|
+
const code = ch.codePointAt(0);
|
|
75
|
+
if (code >= 32 || code === 9 || code === 10 || code === 13) out += ch;
|
|
76
|
+
}
|
|
77
|
+
return out;
|
|
78
|
+
}
|
|
79
|
+
/** Escape a value for use in XML text/attribute content (control chars dropped). */
|
|
80
|
+
function escapeXml(value) {
|
|
81
|
+
return stripXmlControlChars(value).replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """).replace(/'/g, "'");
|
|
82
|
+
}
|
|
83
|
+
/** One-line reason a result failed: its error, else its failing gate labels. */
|
|
84
|
+
function failureMessage(result) {
|
|
85
|
+
if (result.error) return result.error;
|
|
86
|
+
const gates = result.assertions.filter((a) => !a.pass && a.severity === "gate").map((a) => a.detail ? `${a.label} — ${a.detail}` : a.label);
|
|
87
|
+
return gates.length ? gates.join("; ") : "eval failed";
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Render results as JUnit XML for standard CI test reporters: a single
|
|
91
|
+
* `<testsuite name="appkit-agent-evals">` with one `<testcase>` per result.
|
|
92
|
+
* Failures carry a `<failure>` (error or failing-gate summary); skips a
|
|
93
|
+
* `<skipped>`. All attribute/text values are XML-escaped.
|
|
94
|
+
*/
|
|
95
|
+
function formatResultsJUnit(results) {
|
|
96
|
+
const s = summarize(results);
|
|
97
|
+
const lines = [];
|
|
98
|
+
lines.push("<?xml version=\"1.0\" encoding=\"UTF-8\"?>");
|
|
99
|
+
lines.push(`<testsuite name="appkit-agent-evals" tests="${s.total}" failures="${s.failed}" skipped="${s.skipped}">`);
|
|
100
|
+
for (const r of results) {
|
|
101
|
+
const open = ` <testcase name="${escapeXml(r.id)}" classname="agent-eval"`;
|
|
102
|
+
if (r.skipped) {
|
|
103
|
+
lines.push(`${open}>`);
|
|
104
|
+
lines.push(r.skipped.reason ? ` <skipped message="${escapeXml(r.skipped.reason)}"/>` : " <skipped/>");
|
|
105
|
+
lines.push(" </testcase>");
|
|
106
|
+
} else if (!r.passed) {
|
|
107
|
+
const message = failureMessage(r);
|
|
108
|
+
lines.push(`${open}>`);
|
|
109
|
+
lines.push(` <failure message="${escapeXml(message)}"/>`);
|
|
110
|
+
lines.push(" </testcase>");
|
|
111
|
+
} else lines.push(`${open}/>`);
|
|
112
|
+
}
|
|
113
|
+
lines.push("</testsuite>");
|
|
114
|
+
return lines.join("\n");
|
|
115
|
+
}
|
|
54
116
|
|
|
55
117
|
//#endregion
|
|
56
|
-
export { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize };
|
|
118
|
+
export { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize };
|
|
57
119
|
//# sourceMappingURL=report.js.map
|
package/dist/evals/report.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"report.js","names":[],"sources":["../../src/evals/report.ts"],"sourcesContent":["import type { EvalResult } from \"./types\";\n\nexport interface EvalSummary {\n total: number;\n passed: number;\n failed: number;\n skipped: number;\n /** True when no eval failed (skips don't count as failures). */\n allPassed: boolean;\n}\n\nexport function summarize(results: EvalResult[]): EvalSummary {\n let passed = 0;\n let failed = 0;\n let skipped = 0;\n for (const r of results) {\n if (r.skipped) skipped++;\n else if (r.passed) passed++;\n else failed++;\n }\n return {\n total: results.length,\n passed,\n failed,\n skipped,\n allPassed: failed === 0,\n };\n}\n\n/** Status glyph for a single eval result. */\nexport function evalGlyph(result: EvalResult): string {\n if (result.skipped) return \"−\";\n return result.passed ? \"✓\" : \"✗\";\n}\n\n/** The one-line header for a single eval result (no failure detail). */\nexport function formatEvalHeadline(result: EvalResult): string {\n if (result.skipped) {\n return `− ${result.id} (skipped${\n result.skipped.reason ? `: ${result.skipped.reason}` : \"\"\n })`;\n }\n return `${evalGlyph(result)} ${result.id}${\n result.description ? ` — ${result.description}` : \"\"\n }`;\n}\n\n/** Indented detail lines for a failing eval (error + failing assertions). */\nexport function formatEvalDetail(result: EvalResult): string[] {\n const lines: string[] = [];\n if (result.error) lines.push(` error: ${result.error}`);\n for (const a of result.assertions) {\n if (a.pass) continue;\n const tag = a.severity === \"soft\" ? \"soft\" : \"gate\";\n lines.push(` ✗ [${tag}] ${a.label}${a.detail ? ` — ${a.detail}` : \"\"}`);\n }\n return lines;\n}\n\n/** The final PASS/FAIL summary line. */\nexport function formatSummaryLine(results: EvalResult[]): string {\n const s = summarize(results);\n return `${s.allPassed ? \"PASS\" : \"FAIL\"} — ${s.passed} passed, ${s.failed} failed, ${s.skipped} skipped (${s.total} total)`;\n}\n\n/** Render all results as a human-readable console report (non-streaming). */\nexport function formatEvalResults(results: EvalResult[]): string {\n const lines: string[] = [];\n for (const r of results) {\n lines.push(formatEvalHeadline(r));\n lines.push(...formatEvalDetail(r));\n }\n lines.push(\"\");\n lines.push(formatSummaryLine(results));\n return lines.join(\"\\n\");\n}\n"],"mappings":";
|
|
1
|
+
{"version":3,"file":"report.js","names":[],"sources":["../../src/evals/report.ts"],"sourcesContent":["import type { EvalResult } from \"./types\";\n\nexport interface EvalSummary {\n total: number;\n passed: number;\n failed: number;\n skipped: number;\n /** True when no eval failed (skips don't count as failures). */\n allPassed: boolean;\n /** Fraction of scored (non-skipped) evals that passed, 0..1 (1 when none scored). */\n passRate: number;\n}\n\nexport function summarize(results: EvalResult[]): EvalSummary {\n let passed = 0;\n let failed = 0;\n let skipped = 0;\n for (const r of results) {\n if (r.skipped) skipped++;\n else if (r.passed) passed++;\n else failed++;\n }\n const scored = passed + failed;\n return {\n total: results.length,\n passed,\n failed,\n skipped,\n allPassed: failed === 0,\n passRate: scored === 0 ? 1 : passed / scored,\n };\n}\n\n/** Status glyph for a single eval result. */\nexport function evalGlyph(result: EvalResult): string {\n if (result.skipped) return \"−\";\n return result.passed ? \"✓\" : \"✗\";\n}\n\n/** The one-line header for a single eval result (no failure detail). */\nexport function formatEvalHeadline(result: EvalResult): string {\n if (result.skipped) {\n return `− ${result.id} (skipped${\n result.skipped.reason ? `: ${result.skipped.reason}` : \"\"\n })`;\n }\n return `${evalGlyph(result)} ${result.id}${\n result.description ? ` — ${result.description}` : \"\"\n }`;\n}\n\n/** Indented detail lines for a failing eval (error + failing assertions). */\nexport function formatEvalDetail(result: EvalResult): string[] {\n const lines: string[] = [];\n if (result.error) lines.push(` error: ${result.error}`);\n for (const a of result.assertions) {\n if (a.pass) continue;\n const tag = a.severity === \"soft\" ? \"soft\" : \"gate\";\n lines.push(` ✗ [${tag}] ${a.label}${a.detail ? ` — ${a.detail}` : \"\"}`);\n }\n return lines;\n}\n\n/** The final PASS/FAIL summary line. */\nexport function formatSummaryLine(results: EvalResult[]): string {\n const s = summarize(results);\n return `${s.allPassed ? \"PASS\" : \"FAIL\"} — ${s.passed} passed, ${s.failed} failed, ${s.skipped} skipped (${s.total} total)`;\n}\n\n/** Render all results as a human-readable console report (non-streaming). */\nexport function formatEvalResults(results: EvalResult[]): string {\n const lines: string[] = [];\n for (const r of results) {\n lines.push(formatEvalHeadline(r));\n lines.push(...formatEvalDetail(r));\n }\n lines.push(\"\");\n lines.push(formatSummaryLine(results));\n return lines.join(\"\\n\");\n}\n\n/**\n * Render results as a machine-readable JSON report (2-space indented):\n * `{ summary: EvalSummary, results: EvalResult[] }`. Faithful to the types —\n * every field present on a result round-trips.\n */\nexport function formatResultsJson(results: EvalResult[]): string {\n return JSON.stringify({ summary: summarize(results), results }, null, 2);\n}\n\n/**\n * Drop the C0 control chars XML 1.0 forbids even when escaped (except tab, LF,\n * CR) — a raw NUL or ANSI escape would otherwise make the JUnit doc reject on parse.\n */\nfunction stripXmlControlChars(value: string): string {\n // Code-point filter, not a regex: oxlint `no-control-regex` rejects the class.\n let out = \"\";\n for (const ch of value) {\n const code = ch.codePointAt(0) as number;\n if (code >= 0x20 || code === 0x09 || code === 0x0a || code === 0x0d) {\n out += ch;\n }\n }\n return out;\n}\n\n/** Escape a value for use in XML text/attribute content (control chars dropped). */\nfunction escapeXml(value: string): string {\n return stripXmlControlChars(value)\n .replace(/&/g, \"&\")\n .replace(/</g, \"<\")\n .replace(/>/g, \">\")\n .replace(/\"/g, \""\")\n .replace(/'/g, \"'\");\n}\n\n/** One-line reason a result failed: its error, else its failing gate labels. */\nfunction failureMessage(result: EvalResult): string {\n if (result.error) return result.error;\n const gates = result.assertions\n .filter((a) => !a.pass && a.severity === \"gate\")\n .map((a) => (a.detail ? `${a.label} — ${a.detail}` : a.label));\n return gates.length ? gates.join(\"; \") : \"eval failed\";\n}\n\n/**\n * Render results as JUnit XML for standard CI test reporters: a single\n * `<testsuite name=\"appkit-agent-evals\">` with one `<testcase>` per result.\n * Failures carry a `<failure>` (error or failing-gate summary); skips a\n * `<skipped>`. All attribute/text values are XML-escaped.\n */\nexport function formatResultsJUnit(results: EvalResult[]): string {\n const s = summarize(results);\n const lines: string[] = [];\n lines.push('<?xml version=\"1.0\" encoding=\"UTF-8\"?>');\n lines.push(\n `<testsuite name=\"appkit-agent-evals\" tests=\"${s.total}\" failures=\"${s.failed}\" skipped=\"${s.skipped}\">`,\n );\n for (const r of results) {\n const open = ` <testcase name=\"${escapeXml(r.id)}\" classname=\"agent-eval\"`;\n if (r.skipped) {\n lines.push(`${open}>`);\n lines.push(\n r.skipped.reason\n ? ` <skipped message=\"${escapeXml(r.skipped.reason)}\"/>`\n : \" <skipped/>\",\n );\n lines.push(\" </testcase>\");\n } else if (!r.passed) {\n const message = failureMessage(r);\n lines.push(`${open}>`);\n lines.push(` <failure message=\"${escapeXml(message)}\"/>`);\n lines.push(\" </testcase>\");\n } else {\n lines.push(`${open}/>`);\n }\n }\n lines.push(\"</testsuite>\");\n return lines.join(\"\\n\");\n}\n"],"mappings":";AAaA,SAAgB,UAAU,SAAoC;CAC5D,IAAI,SAAS;CACb,IAAI,SAAS;CACb,IAAI,UAAU;AACd,MAAK,MAAM,KAAK,QACd,KAAI,EAAE,QAAS;UACN,EAAE,OAAQ;KACd;CAEP,MAAM,SAAS,SAAS;AACxB,QAAO;EACL,OAAO,QAAQ;EACf;EACA;EACA;EACA,WAAW,WAAW;EACtB,UAAU,WAAW,IAAI,IAAI,SAAS;EACvC;;;AAIH,SAAgB,UAAU,QAA4B;AACpD,KAAI,OAAO,QAAS,QAAO;AAC3B,QAAO,OAAO,SAAS,MAAM;;;AAI/B,SAAgB,mBAAmB,QAA4B;AAC7D,KAAI,OAAO,QACT,QAAO,KAAK,OAAO,GAAG,WACpB,OAAO,QAAQ,SAAS,KAAK,OAAO,QAAQ,WAAW,GACxD;AAEH,QAAO,GAAG,UAAU,OAAO,CAAC,GAAG,OAAO,KACpC,OAAO,cAAc,MAAM,OAAO,gBAAgB;;;AAKtD,SAAgB,iBAAiB,QAA8B;CAC7D,MAAM,QAAkB,EAAE;AAC1B,KAAI,OAAO,MAAO,OAAM,KAAK,cAAc,OAAO,QAAQ;AAC1D,MAAK,MAAM,KAAK,OAAO,YAAY;AACjC,MAAI,EAAE,KAAM;EACZ,MAAM,MAAM,EAAE,aAAa,SAAS,SAAS;AAC7C,QAAM,KAAK,UAAU,IAAI,IAAI,EAAE,QAAQ,EAAE,SAAS,MAAM,EAAE,WAAW,KAAK;;AAE5E,QAAO;;;AAIT,SAAgB,kBAAkB,SAA+B;CAC/D,MAAM,IAAI,UAAU,QAAQ;AAC5B,QAAO,GAAG,EAAE,YAAY,SAAS,OAAO,KAAK,EAAE,OAAO,WAAW,EAAE,OAAO,WAAW,EAAE,QAAQ,YAAY,EAAE,MAAM;;;AAIrH,SAAgB,kBAAkB,SAA+B;CAC/D,MAAM,QAAkB,EAAE;AAC1B,MAAK,MAAM,KAAK,SAAS;AACvB,QAAM,KAAK,mBAAmB,EAAE,CAAC;AACjC,QAAM,KAAK,GAAG,iBAAiB,EAAE,CAAC;;AAEpC,OAAM,KAAK,GAAG;AACd,OAAM,KAAK,kBAAkB,QAAQ,CAAC;AACtC,QAAO,MAAM,KAAK,KAAK;;;;;;;AAQzB,SAAgB,kBAAkB,SAA+B;AAC/D,QAAO,KAAK,UAAU;EAAE,SAAS,UAAU,QAAQ;EAAE;EAAS,EAAE,MAAM,EAAE;;;;;;AAO1E,SAAS,qBAAqB,OAAuB;CAEnD,IAAI,MAAM;AACV,MAAK,MAAM,MAAM,OAAO;EACtB,MAAM,OAAO,GAAG,YAAY,EAAE;AAC9B,MAAI,QAAQ,MAAQ,SAAS,KAAQ,SAAS,MAAQ,SAAS,GAC7D,QAAO;;AAGX,QAAO;;;AAIT,SAAS,UAAU,OAAuB;AACxC,QAAO,qBAAqB,MAAM,CAC/B,QAAQ,MAAM,QAAQ,CACtB,QAAQ,MAAM,OAAO,CACrB,QAAQ,MAAM,OAAO,CACrB,QAAQ,MAAM,SAAS,CACvB,QAAQ,MAAM,SAAS;;;AAI5B,SAAS,eAAe,QAA4B;AAClD,KAAI,OAAO,MAAO,QAAO,OAAO;CAChC,MAAM,QAAQ,OAAO,WAClB,QAAQ,MAAM,CAAC,EAAE,QAAQ,EAAE,aAAa,OAAO,CAC/C,KAAK,MAAO,EAAE,SAAS,GAAG,EAAE,MAAM,KAAK,EAAE,WAAW,EAAE,MAAO;AAChE,QAAO,MAAM,SAAS,MAAM,KAAK,KAAK,GAAG;;;;;;;;AAS3C,SAAgB,mBAAmB,SAA+B;CAChE,MAAM,IAAI,UAAU,QAAQ;CAC5B,MAAM,QAAkB,EAAE;AAC1B,OAAM,KAAK,6CAAyC;AACpD,OAAM,KACJ,+CAA+C,EAAE,MAAM,cAAc,EAAE,OAAO,aAAa,EAAE,QAAQ,IACtG;AACD,MAAK,MAAM,KAAK,SAAS;EACvB,MAAM,OAAO,qBAAqB,UAAU,EAAE,GAAG,CAAC;AAClD,MAAI,EAAE,SAAS;AACb,SAAM,KAAK,GAAG,KAAK,GAAG;AACtB,SAAM,KACJ,EAAE,QAAQ,SACN,yBAAyB,UAAU,EAAE,QAAQ,OAAO,CAAC,OACrD,iBACL;AACD,SAAM,KAAK,gBAAgB;aAClB,CAAC,EAAE,QAAQ;GACpB,MAAM,UAAU,eAAe,EAAE;AACjC,SAAM,KAAK,GAAG,KAAK,GAAG;AACtB,SAAM,KAAK,yBAAyB,UAAU,QAAQ,CAAC,KAAK;AAC5D,SAAM,KAAK,gBAAgB;QAE3B,OAAM,KAAK,GAAG,KAAK,IAAI;;AAG3B,OAAM,KAAK,eAAe;AAC1B,QAAO,MAAM,KAAK,KAAK"}
|
package/dist/evals/run-eval.d.ts
CHANGED
|
@@ -11,6 +11,11 @@ interface RunEvalOptions {
|
|
|
11
11
|
strict?: boolean;
|
|
12
12
|
/** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */
|
|
13
13
|
row?: DatasetRow;
|
|
14
|
+
/**
|
|
15
|
+
* Runner-level default per-eval timeout (ms). `def.timeoutMs` wins over this;
|
|
16
|
+
* when both are unset the eval runs unbounded (current behavior).
|
|
17
|
+
*/
|
|
18
|
+
timeoutMs?: number;
|
|
14
19
|
}
|
|
15
20
|
/**
|
|
16
21
|
* Run a single eval against a driver. Never throws for assertion or agent
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-eval.d.ts","names":[],"sources":["../../src/evals/run-eval.ts"],"mappings":";;;;
|
|
1
|
+
{"version":3,"file":"run-eval.d.ts","names":[],"sources":["../../src/evals/run-eval.ts"],"mappings":";;;;UAoDiB,cAAA;;EAEf,EAAA;EAF6B;EAI7B,MAAA,EAAQ,UAAA;EAIQ;EAFhB,MAAA;EAFA;EAIA,GAAA,GAAM,UAAA;EAFN;;;;EAOA,SAAA;AAAA;AAQF;;;;;AAAA,iBAAsB,OAAA,CACpB,GAAA,EAAK,cAAA,EACL,OAAA,EAAS,cAAA,GACR,OAAA,CAAQ,UAAA"}
|
package/dist/evals/run-eval.js
CHANGED
|
@@ -12,6 +12,22 @@ var SkipSignal = class extends Error {
|
|
|
12
12
|
}
|
|
13
13
|
};
|
|
14
14
|
/**
|
|
15
|
+
* Deep partial match: every key in `expected` is present in `actual` and equal,
|
|
16
|
+
* recursing into nested plain objects so extra actual keys are ignored. Arrays
|
|
17
|
+
* match element-for-element (same length, deep-equal items).
|
|
18
|
+
*/
|
|
19
|
+
function deepContains(actual, expected) {
|
|
20
|
+
if (Array.isArray(expected)) return Array.isArray(actual) && actual.length === expected.length && expected.every((item, i) => deepContains(actual[i], item));
|
|
21
|
+
if (isPlainObject(expected)) {
|
|
22
|
+
if (!isPlainObject(actual)) return false;
|
|
23
|
+
return Object.keys(expected).every((key) => Object.hasOwn(actual, key) && deepContains(actual[key], expected[key]));
|
|
24
|
+
}
|
|
25
|
+
return actual === expected;
|
|
26
|
+
}
|
|
27
|
+
function isPlainObject(value) {
|
|
28
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
15
31
|
* Run a single eval against a driver. Never throws for assertion or agent
|
|
16
32
|
* failures — those become a non-passing {@link EvalResult}. Only a malformed
|
|
17
33
|
* eval definition surfaces as `result.error`.
|
|
@@ -21,9 +37,12 @@ async function runEval(def, options) {
|
|
|
21
37
|
let reply = "";
|
|
22
38
|
let lastInput = "";
|
|
23
39
|
let toolCalls = [];
|
|
40
|
+
let toolCallDetails = [];
|
|
24
41
|
let sessionId;
|
|
25
42
|
let lastTraceId;
|
|
26
43
|
let lastSucceeded = false;
|
|
44
|
+
let turnFailed = false;
|
|
45
|
+
const controller = new AbortController();
|
|
27
46
|
const record = (label, pass, score, detail) => {
|
|
28
47
|
const result = {
|
|
29
48
|
label,
|
|
@@ -53,11 +72,13 @@ async function runEval(def, options) {
|
|
|
53
72
|
const t = {
|
|
54
73
|
async send(message) {
|
|
55
74
|
lastInput = message;
|
|
56
|
-
const r = await options.driver.send(message);
|
|
75
|
+
const r = await options.driver.send(message, { signal: controller.signal });
|
|
57
76
|
reply = r.reply;
|
|
58
77
|
toolCalls = r.toolCalls;
|
|
78
|
+
toolCallDetails = r.toolCallDetails;
|
|
59
79
|
sessionId = r.sessionId;
|
|
60
80
|
lastSucceeded = r.succeeded;
|
|
81
|
+
if (!r.succeeded) turnFailed = true;
|
|
61
82
|
if (r.traceId) lastTraceId = r.traceId;
|
|
62
83
|
},
|
|
63
84
|
reset() {
|
|
@@ -84,6 +105,12 @@ async function runEval(def, options) {
|
|
|
84
105
|
calledTool(name) {
|
|
85
106
|
return record(`calledTool(${name})`, toolCalls.includes(name), void 0, `expected tool "${name}" to be called (called: ${toolCalls.length ? toolCalls.join(", ") : "none"})`);
|
|
86
107
|
},
|
|
108
|
+
calledToolWith(name, expected) {
|
|
109
|
+
const matching = toolCallDetails.filter((c) => c.name === name);
|
|
110
|
+
const pass = matching.some((c) => deepContains(c.args, expected));
|
|
111
|
+
const seen = matching.length ? matching.map((c) => `{${Object.keys(c.args).sort().join(", ")}}`).join(", ") : "not called";
|
|
112
|
+
return record(`calledToolWith(${name})`, pass, void 0, `expected tool "${name}" to be called with ${JSON.stringify(expected)} (arg keys seen: ${seen})`);
|
|
113
|
+
},
|
|
87
114
|
check(value, matcher) {
|
|
88
115
|
const m = matcher(value);
|
|
89
116
|
return record("check", m.pass, m.score, m.detail);
|
|
@@ -117,8 +144,19 @@ async function runEval(def, options) {
|
|
|
117
144
|
throw new SkipSignal(reason);
|
|
118
145
|
}
|
|
119
146
|
};
|
|
147
|
+
const timeoutMs = def.timeoutMs ?? options.timeoutMs;
|
|
148
|
+
let timer;
|
|
120
149
|
try {
|
|
121
|
-
await def.test(t);
|
|
150
|
+
if (timeoutMs === void 0) await def.test(t);
|
|
151
|
+
else {
|
|
152
|
+
const timeout = new Promise((_, reject) => {
|
|
153
|
+
timer = setTimeout(() => {
|
|
154
|
+
controller.abort();
|
|
155
|
+
reject(/* @__PURE__ */ new Error(`eval timed out after ${timeoutMs}ms`));
|
|
156
|
+
}, timeoutMs);
|
|
157
|
+
});
|
|
158
|
+
await Promise.race([Promise.resolve(def.test(t)), timeout]);
|
|
159
|
+
}
|
|
122
160
|
} catch (err) {
|
|
123
161
|
if (err instanceof SkipSignal) return {
|
|
124
162
|
id: options.id,
|
|
@@ -136,6 +174,8 @@ async function runEval(def, options) {
|
|
|
136
174
|
error: err instanceof Error ? err.message : String(err),
|
|
137
175
|
traceId: lastTraceId
|
|
138
176
|
};
|
|
177
|
+
} finally {
|
|
178
|
+
if (timer) clearTimeout(timer);
|
|
139
179
|
}
|
|
140
180
|
const passed = assertions.every((a) => a.pass || a.severity === "soft" && !options.strict);
|
|
141
181
|
return {
|
|
@@ -143,7 +183,8 @@ async function runEval(def, options) {
|
|
|
143
183
|
description: def.description,
|
|
144
184
|
assertions,
|
|
145
185
|
passed,
|
|
146
|
-
traceId: lastTraceId
|
|
186
|
+
traceId: lastTraceId,
|
|
187
|
+
infraFailure: turnFailed && !passed || void 0
|
|
147
188
|
};
|
|
148
189
|
}
|
|
149
190
|
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import type { DatasetRow } from \"./dataset\";\nimport { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */\n row?: DatasetRow;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n // Re-threshold the score; keep the current severity (gate unless the\n // caller also chained `.soft()`).\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and gate by default (a miss fails the eval,\n // like other assertions). The caller chains `.atLeast(n)` to change the pass\n // threshold, or `.soft()` to demote to a tracked-only metric.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message);\n reply = r.reply;\n toolCalls = r.toolCalls;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (r.traceId) lastTraceId = r.traceId;\n },\n reset() {\n options.driver.reset?.();\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n get input() {\n return options.row?.inputs ?? {};\n },\n get expected() {\n return options.row?.expectations;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n try {\n await def.test(t);\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n };\n}\n"],"mappings":";;;;AAaA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAoBhB,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AAGzB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAMT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU;CAEnE,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,QAAQ;AAC5C,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,QAAQ;AACN,WAAQ,OAAO,SAAS;;EAE1B,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,QAAQ;AACV,UAAO,QAAQ,KAAK,UAAU,EAAE;;EAElC,IAAI,WAAW;AACb,UAAO,QAAQ,KAAK;;EAEtB,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;AAED,KAAI;AACF,QAAM,IAAI,KAAK,EAAE;UACV,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;;CAGH,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EACV"}
|
|
1
|
+
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import type { DatasetRow } from \"./dataset\";\nimport { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n DriveResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\n/**\n * Deep partial match: every key in `expected` is present in `actual` and equal,\n * recursing into nested plain objects so extra actual keys are ignored. Arrays\n * match element-for-element (same length, deep-equal items).\n */\nfunction deepContains(actual: unknown, expected: unknown): boolean {\n if (Array.isArray(expected)) {\n return (\n Array.isArray(actual) &&\n actual.length === expected.length &&\n expected.every((item, i) => deepContains(actual[i], item))\n );\n }\n if (isPlainObject(expected)) {\n if (!isPlainObject(actual)) return false;\n // Require the key present: an expected `undefined` must not match an omitted key.\n return Object.keys(expected).every(\n (key) =>\n Object.hasOwn(actual, key) && deepContains(actual[key], expected[key]),\n );\n }\n return actual === expected;\n}\n\nfunction isPlainObject(value: unknown): value is Record<string, unknown> {\n return typeof value === \"object\" && value !== null && !Array.isArray(value);\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */\n row?: DatasetRow;\n /**\n * Runner-level default per-eval timeout (ms). `def.timeoutMs` wins over this;\n * when both are unset the eval runs unbounded (current behavior).\n */\n timeoutMs?: number;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let toolCallDetails: DriveResult[\"toolCallDetails\"] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n // Any transport/agent turn failure (driver `succeeded: false`) → `infraFailure`, for retry.\n let turnFailed = false;\n // Aborted on per-eval timeout to cancel the in-flight turn (no leaked stream).\n const controller = new AbortController();\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n // Re-threshold the score; keep the current severity (gate unless the\n // caller also chained `.soft()`).\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and gate by default (a miss fails the eval,\n // like other assertions). The caller chains `.atLeast(n)` to change the pass\n // threshold, or `.soft()` to demote to a tracked-only metric.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message, {\n signal: controller.signal,\n });\n reply = r.reply;\n toolCalls = r.toolCalls;\n toolCallDetails = r.toolCallDetails;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (!r.succeeded) turnFailed = true;\n if (r.traceId) lastTraceId = r.traceId;\n },\n reset() {\n options.driver.reset?.();\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n get input() {\n return options.row?.inputs ?? {};\n },\n get expected() {\n return options.row?.expectations;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n calledToolWith(name, expected) {\n const matching = toolCallDetails.filter((c) => c.name === name);\n const pass = matching.some((c) => deepContains(c.args, expected));\n // Keys only, never values: actual args may hold PII/secrets and are persisted to reports (CWE-532).\n const seen = matching.length\n ? matching\n .map((c) => `{${Object.keys(c.args).sort().join(\", \")}}`)\n .join(\", \")\n : \"not called\";\n return record(\n `calledToolWith(${name})`,\n pass,\n undefined,\n `expected tool \"${name}\" to be called with ${JSON.stringify(\n expected,\n )} (arg keys seen: ${seen})`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n // `def.timeoutMs` (per-eval) wins over the runner default; when both are\n // unset the eval runs unbounded (undefined = no timeout).\n const timeoutMs = def.timeoutMs ?? options.timeoutMs;\n let timer: ReturnType<typeof setTimeout> | undefined;\n\n try {\n if (timeoutMs === undefined) {\n await def.test(t);\n } else {\n // Race the test against the timeout; on elapse, abort the turn and settle\n // non-passing. Only the driver turn cancels — non-driver hangs (judge, sleep) run on.\n const timeout = new Promise<never>((_, reject) => {\n timer = setTimeout(() => {\n controller.abort();\n reject(new Error(`eval timed out after ${timeoutMs}ms`));\n }, timeoutMs);\n });\n await Promise.race([Promise.resolve(def.test(t)), timeout]);\n }\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n } finally {\n if (timer) clearTimeout(timer);\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n // Only a *failing* eval whose turn broke at the transport/agent level is a\n // retryable infra flake; a passing eval (or a pure assertion mismatch, where\n // the turn itself succeeded) is real signal and must not be retried.\n infraFailure: (turnFailed && !passed) || undefined,\n };\n}\n"],"mappings":";;;;AAcA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAShB,SAAS,aAAa,QAAiB,UAA4B;AACjE,KAAI,MAAM,QAAQ,SAAS,CACzB,QACE,MAAM,QAAQ,OAAO,IACrB,OAAO,WAAW,SAAS,UAC3B,SAAS,OAAO,MAAM,MAAM,aAAa,OAAO,IAAI,KAAK,CAAC;AAG9D,KAAI,cAAc,SAAS,EAAE;AAC3B,MAAI,CAAC,cAAc,OAAO,CAAE,QAAO;AAEnC,SAAO,OAAO,KAAK,SAAS,CAAC,OAC1B,QACC,OAAO,OAAO,QAAQ,IAAI,IAAI,aAAa,OAAO,MAAM,SAAS,KAAK,CACzE;;AAEH,QAAO,WAAW;;AAGpB,SAAS,cAAc,OAAkD;AACvE,QAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,MAAM;;;;;;;AAwB7E,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI,kBAAkD,EAAE;CACxD,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,IAAI,aAAa;CAEjB,MAAM,aAAa,IAAI,iBAAiB;CAExC,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AAGzB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAMT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU;CAEnE,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,SAAS,EAC3C,QAAQ,WAAW,QACpB,CAAC;AACF,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,qBAAkB,EAAE;AACpB,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,CAAC,EAAE,UAAW,cAAa;AAC/B,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,QAAQ;AACN,WAAQ,OAAO,SAAS;;EAE1B,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,QAAQ;AACV,UAAO,QAAQ,KAAK,UAAU,EAAE;;EAElC,IAAI,WAAW;AACb,UAAO,QAAQ,KAAK;;EAEtB,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,eAAe,MAAM,UAAU;GAC7B,MAAM,WAAW,gBAAgB,QAAQ,MAAM,EAAE,SAAS,KAAK;GAC/D,MAAM,OAAO,SAAS,MAAM,MAAM,aAAa,EAAE,MAAM,SAAS,CAAC;GAEjE,MAAM,OAAO,SAAS,SAClB,SACG,KAAK,MAAM,IAAI,OAAO,KAAK,EAAE,KAAK,CAAC,MAAM,CAAC,KAAK,KAAK,CAAC,GAAG,CACxD,KAAK,KAAK,GACb;AACJ,UAAO,OACL,kBAAkB,KAAK,IACvB,MACA,QACA,kBAAkB,KAAK,sBAAsB,KAAK,UAChD,SACD,CAAC,mBAAmB,KAAK,GAC3B;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;CAID,MAAM,YAAY,IAAI,aAAa,QAAQ;CAC3C,IAAI;AAEJ,KAAI;AACF,MAAI,cAAc,OAChB,OAAM,IAAI,KAAK,EAAE;OACZ;GAGL,MAAM,UAAU,IAAI,SAAgB,GAAG,WAAW;AAChD,YAAQ,iBAAiB;AACvB,gBAAW,OAAO;AAClB,4BAAO,IAAI,MAAM,wBAAwB,UAAU,IAAI,CAAC;OACvD,UAAU;KACb;AACF,SAAM,QAAQ,KAAK,CAAC,QAAQ,QAAQ,IAAI,KAAK,EAAE,CAAC,EAAE,QAAQ,CAAC;;UAEtD,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;WACO;AACR,MAAI,MAAO,cAAa,MAAM;;CAGhC,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EAIT,cAAe,cAAc,CAAC,UAAW;EAC1C"}
|
|
@@ -12,12 +12,15 @@ interface RunEvalsOptions {
|
|
|
12
12
|
baseUrl: string;
|
|
13
13
|
/** Substring filter on `<agent>/<id>` (or an exact agent id). */
|
|
14
14
|
filter?: string;
|
|
15
|
+
/**
|
|
16
|
+
* Only run evals whose `tags` intersect this list. Empty/undefined runs all.
|
|
17
|
+
* Tags live on the eval def, so filtering happens after each file is loaded.
|
|
18
|
+
*/
|
|
19
|
+
tags?: string[];
|
|
15
20
|
/** Soft assertion failures also fail the eval. */
|
|
16
21
|
strict?: boolean;
|
|
17
22
|
/** Extra request headers for the driver (e.g. auth for a deployed app). */
|
|
18
23
|
headers?: Record<string, string>;
|
|
19
|
-
/** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */
|
|
20
|
-
timeoutMs?: number;
|
|
21
24
|
/**
|
|
22
25
|
* Max evals to drive concurrently. Each eval opens one stream to the app as
|
|
23
26
|
* the same user, so keep this at or below the app's
|
|
@@ -54,6 +57,18 @@ interface RunEvalsOptions {
|
|
|
54
57
|
warehouseId?: string;
|
|
55
58
|
/** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */
|
|
56
59
|
now?: number;
|
|
60
|
+
/**
|
|
61
|
+
* Default per-eval timeout (ms): `runEval` races the whole test against it and
|
|
62
|
+
* it also caps each driver turn. A per-eval `def.timeoutMs` overrides it, and
|
|
63
|
+
* it wins over an agent's `evals.config.ts` `timeoutMs`. Unbounded when unset.
|
|
64
|
+
*/
|
|
65
|
+
timeoutMs?: number;
|
|
66
|
+
/**
|
|
67
|
+
* Re-run an eval up to this many extra times when it fails on infrastructure —
|
|
68
|
+
* a thrown error/timeout (`result.error`) or a transport/agent turn failure
|
|
69
|
+
* (`result.infraFailure`). Assertion failures are never retried. Defaults to `0`.
|
|
70
|
+
*/
|
|
71
|
+
retries?: number;
|
|
57
72
|
/** Progress callback, invoked as evals are discovered, started, and finished. */
|
|
58
73
|
onEvent?: (event: EvalProgress) => void;
|
|
59
74
|
}
|
|
@@ -83,6 +98,20 @@ interface EvalRunSummary {
|
|
|
83
98
|
finish: FinishOutcome;
|
|
84
99
|
};
|
|
85
100
|
}
|
|
101
|
+
/**
|
|
102
|
+
* Run `attempt` up to `1 + retries` times, stopping as soon as it returns a
|
|
103
|
+
* result that is neither a thrown error / per-eval timeout (`error`) nor a
|
|
104
|
+
* transport/agent turn failure (`infraFailure`). Assertion failures set
|
|
105
|
+
* neither, so a failed-but-completed eval is returned on the first try and
|
|
106
|
+
* never retried. Returns the last result when every attempt failed on infra.
|
|
107
|
+
*
|
|
108
|
+
* Between attempts it waits a full-jittered exponential backoff (infra flakes
|
|
109
|
+
* are overload-correlated). `retries` is coerced to a finite non-negative
|
|
110
|
+
* integer; `baseDelayMs: 0` disables the wait (tests).
|
|
111
|
+
*/
|
|
112
|
+
declare function runWithRetries(retries: number, attempt: (attemptNumber: number) => Promise<EvalResult>, options?: {
|
|
113
|
+
baseDelayMs?: number;
|
|
114
|
+
}): Promise<EvalResult>;
|
|
86
115
|
/**
|
|
87
116
|
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
88
117
|
* the agents on a running app. Never throws for an individual eval — load/run
|
|
@@ -90,5 +119,5 @@ interface EvalRunSummary {
|
|
|
90
119
|
*/
|
|
91
120
|
declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
|
|
92
121
|
//#endregion
|
|
93
|
-
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir };
|
|
122
|
+
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries };
|
|
94
123
|
//# sourceMappingURL=run-evals.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;
|
|
1
|
+
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAmBiB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAWU;EATV,MAAA;EAwDkB;;;;EAnDlB,IAAA;EALA;EAOA,MAAA;EAAA;EAEA,OAAA,GAAU,MAAA;EAAA;;;;;;EAOV,WAAA;EAiBA;;;;;EAXA,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UA6BF;IA3BE,cAAA;EAAA;EA6BS;;;AAGb;EA1BE,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EA4BnC;;;;EAvBJ,eAAA,GAAkB,eAAA;EAwB4B;EAtB9C,WAAA;EAuBoB;EArBpB,GAAA;EAqBwC;;;;AAE1C;EAjBE,SAAA;;;;;;EAMA,OAAA;EAYA;EAVA,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EA2OmC;EAzO5C,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;;;;;;;iBAuOrC,cAAA,CACpB,OAAA,UACA,OAAA,GAAU,aAAA,aAA0B,OAAA,CAAQ,UAAA,GAC5C,OAAA;EAAW,WAAA;AAAA,IACV,OAAA,CAAQ,UAAA;;;;;;iBA8GW,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
|