@databricks/appkit 0.71.0 → 0.73.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +63 -0
- package/NOTICE.md +3 -2
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +18 -3
- package/dist/beta.js +14 -1
- package/dist/cli/commands/agent/eval.js +120 -0
- package/dist/cli/commands/agent/eval.js.map +1 -0
- package/dist/cli/commands/agent/index.js +18 -0
- package/dist/cli/commands/agent/index.js.map +1 -0
- package/dist/cli/index.js +2 -0
- package/dist/cli/index.js.map +1 -1
- package/dist/connectors/index.js +2 -0
- package/dist/connectors/mlflow/auth.d.ts +28 -0
- package/dist/connectors/mlflow/auth.d.ts.map +1 -0
- package/dist/connectors/mlflow/auth.js +70 -0
- package/dist/connectors/mlflow/auth.js.map +1 -0
- package/dist/connectors/mlflow/client.d.ts +51 -0
- package/dist/connectors/mlflow/client.d.ts.map +1 -0
- package/dist/connectors/mlflow/client.js +93 -0
- package/dist/connectors/mlflow/client.js.map +1 -0
- package/dist/connectors/mlflow/index.d.ts +2 -0
- package/dist/database/errors.js +15 -5
- package/dist/database/errors.js.map +1 -1
- package/dist/database/runtime/data-path.d.ts +7 -0
- package/dist/database/runtime/data-path.d.ts.map +1 -0
- package/dist/database/runtime/data-path.js.map +1 -1
- package/dist/database/runtime/engine/drizzle-data-path.js +7 -5
- package/dist/database/runtime/engine/drizzle-data-path.js.map +1 -1
- package/dist/database/schema-builder/define-schema.d.ts +1 -1
- package/dist/database/schema-builder/define-schema.js +1 -1
- package/dist/database/schema-builder/define-schema.js.map +1 -1
- package/dist/errors/database-validation.d.ts +23 -0
- package/dist/errors/database-validation.d.ts.map +1 -0
- package/dist/errors/database-validation.js +24 -0
- package/dist/errors/database-validation.js.map +1 -0
- package/dist/errors/index.js +1 -0
- package/dist/evals/dataset.d.ts +36 -0
- package/dist/evals/dataset.d.ts.map +1 -0
- package/dist/evals/dataset.js +36 -0
- package/dist/evals/dataset.js.map +1 -0
- package/dist/evals/define-eval.d.ts +26 -0
- package/dist/evals/define-eval.d.ts.map +1 -0
- package/dist/evals/define-eval.js +28 -0
- package/dist/evals/define-eval.js.map +1 -0
- package/dist/evals/discover.d.ts +20 -0
- package/dist/evals/discover.d.ts.map +1 -0
- package/dist/evals/discover.js +49 -0
- package/dist/evals/discover.js.map +1 -0
- package/dist/evals/http-driver.d.ts +33 -0
- package/dist/evals/http-driver.d.ts.map +1 -0
- package/dist/evals/http-driver.js +123 -0
- package/dist/evals/http-driver.js.map +1 -0
- package/dist/evals/index.d.ts +14 -0
- package/dist/evals/index.js +14 -0
- package/dist/evals/judge.d.ts +27 -0
- package/dist/evals/judge.d.ts.map +1 -0
- package/dist/evals/judge.js +77 -0
- package/dist/evals/judge.js.map +1 -0
- package/dist/evals/matchers.d.ts +12 -0
- package/dist/evals/matchers.d.ts.map +1 -0
- package/dist/evals/matchers.js +26 -0
- package/dist/evals/matchers.js.map +1 -0
- package/dist/evals/mlflow-report.d.ts +37 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -0
- package/dist/evals/mlflow-report.js +161 -0
- package/dist/evals/mlflow-report.js.map +1 -0
- package/dist/evals/mlflow-run.d.ts +13 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -0
- package/dist/evals/mlflow-run.js +101 -0
- package/dist/evals/mlflow-run.js.map +1 -0
- package/dist/evals/pool.js +24 -0
- package/dist/evals/pool.js.map +1 -0
- package/dist/evals/report.d.ts +25 -0
- package/dist/evals/report.d.ts.map +1 -0
- package/dist/evals/report.js +57 -0
- package/dist/evals/report.js.map +1 -0
- package/dist/evals/run-eval.d.ts +23 -0
- package/dist/evals/run-eval.d.ts.map +1 -0
- package/dist/evals/run-eval.js +152 -0
- package/dist/evals/run-eval.js.map +1 -0
- package/dist/evals/run-evals.d.ts +94 -0
- package/dist/evals/run-evals.d.ts.map +1 -0
- package/dist/evals/run-evals.js +257 -0
- package/dist/evals/run-evals.js.map +1 -0
- package/dist/evals/types.d.ts +163 -0
- package/dist/evals/types.d.ts.map +1 -0
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/plugin/plugin.d.ts.map +1 -1
- package/dist/plugin/plugin.js +1 -1
- package/dist/plugin/plugin.js.map +1 -1
- package/dist/plugins/agents/agents.js +1 -1
- package/dist/plugins/database/crud/contract.js +17 -8
- package/dist/plugins/database/crud/contract.js.map +1 -1
- package/dist/plugins/database/crud/exposure.js +63 -22
- package/dist/plugins/database/crud/exposure.js.map +1 -1
- package/dist/plugins/database/crud/request.js +50 -0
- package/dist/plugins/database/crud/request.js.map +1 -0
- package/dist/plugins/database/crud/response.js +77 -0
- package/dist/plugins/database/crud/response.js.map +1 -0
- package/dist/plugins/database/crud/routes.js +71 -52
- package/dist/plugins/database/crud/routes.js.map +1 -1
- package/dist/plugins/database/database.d.ts +6 -4
- package/dist/plugins/database/database.d.ts.map +1 -1
- package/dist/plugins/database/database.js +46 -16
- package/dist/plugins/database/database.js.map +1 -1
- package/dist/plugins/database/defaults.js +5 -1
- package/dist/plugins/database/defaults.js.map +1 -1
- package/dist/plugins/database/entity-client.js +143 -10
- package/dist/plugins/database/entity-client.js.map +1 -1
- package/dist/plugins/database/entity-types.d.ts +1 -1
- package/dist/plugins/database/hooks.d.ts +38 -0
- package/dist/plugins/database/hooks.d.ts.map +1 -0
- package/dist/plugins/database/index.d.ts +3 -2
- package/dist/plugins/database/lifecycle.js +67 -28
- package/dist/plugins/database/lifecycle.js.map +1 -1
- package/dist/plugins/database/scope.js +58 -0
- package/dist/plugins/database/scope.js.map +1 -0
- package/dist/plugins/database/types.d.ts +40 -12
- package/dist/plugins/database/types.d.ts.map +1 -1
- package/docs/api/appkit/Class.AppKitError.md +1 -0
- package/docs/api/appkit/Class.DatabaseValidationError.md +191 -0
- package/docs/api/appkit/Class.MlflowClient.md +103 -0
- package/docs/api/appkit/Function.buildAssessments.md +16 -0
- package/docs/api/appkit/Function.configureJudge.md +18 -0
- package/docs/api/appkit/Function.createHttpDriver.md +18 -0
- package/docs/api/appkit/Function.defineEval.md +35 -0
- package/docs/api/appkit/Function.defineSchema.md +1 -1
- package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
- package/docs/api/appkit/Function.equals.md +18 -0
- package/docs/api/appkit/Function.evalGlyph.md +18 -0
- package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
- package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
- package/docs/api/appkit/Function.formatEvalResults.md +18 -0
- package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
- package/docs/api/appkit/Function.includes.md +18 -0
- package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
- package/docs/api/appkit/Function.matches.md +18 -0
- package/docs/api/appkit/Function.normalizeHost.md +18 -0
- package/docs/api/appkit/Function.readEvalDataset.md +21 -0
- package/docs/api/appkit/Function.reportToMlflow.md +23 -0
- package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
- package/docs/api/appkit/Function.resolveWorkspaceClient.md +18 -0
- package/docs/api/appkit/Function.runEval.md +19 -0
- package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
- package/docs/api/appkit/Function.summarize.md +16 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
- package/docs/api/appkit/Interface.AssertionResult.md +48 -0
- package/docs/api/appkit/Interface.Assessment.md +83 -0
- package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
- package/docs/api/appkit/Interface.DatabaseValidationIssue.md +21 -0
- package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
- package/docs/api/appkit/Interface.DatasetRow.md +21 -0
- package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
- package/docs/api/appkit/Interface.DriveResult.md +58 -0
- package/docs/api/appkit/Interface.EntityMutationHooks.md +173 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +74 -0
- package/docs/api/appkit/Interface.EvalDriver.md +37 -0
- package/docs/api/appkit/Interface.EvalResult.md +83 -0
- package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
- package/docs/api/appkit/Interface.EvalSummary.md +48 -0
- package/docs/api/appkit/Interface.HookApp.md +12 -0
- package/docs/api/appkit/Interface.HookContext.md +21 -0
- package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
- package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
- package/docs/api/appkit/Interface.JudgeScore.md +21 -0
- package/docs/api/appkit/Interface.MatchResult.md +34 -0
- package/docs/api/appkit/Interface.PostResult.md +30 -0
- package/docs/api/appkit/Interface.ReadEvalDatasetOptions.md +34 -0
- package/docs/api/appkit/Interface.ReadSerializerContext.md +21 -0
- package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
- package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +45 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +214 -0
- package/docs/api/appkit/Interface.TestContext.md +245 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiConfig.md +53 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md +8 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md +49 -0
- package/docs/api/appkit/TypeAlias.DatabaseExports.md +3 -3
- package/docs/api/appkit/TypeAlias.EntityHooks.md +25 -0
- package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
- package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +16 -5
- package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
- package/docs/api/appkit/TypeAlias.ReadSerializer.md +19 -0
- package/docs/api/appkit/TypeAlias.Severity.md +8 -0
- package/docs/api/appkit/TypeAlias.TransactionClient.md +19 -0
- package/docs/api/appkit.md +157 -95
- package/docs/plugins/database.md +144 -0
- package/llms.txt +63 -0
- package/package.json +3 -2
- package/sbom.cdx.json +1 -1
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"report.js","names":[],"sources":["../../src/evals/report.ts"],"sourcesContent":["import type { EvalResult } from \"./types\";\n\nexport interface EvalSummary {\n total: number;\n passed: number;\n failed: number;\n skipped: number;\n /** True when no eval failed (skips don't count as failures). */\n allPassed: boolean;\n}\n\nexport function summarize(results: EvalResult[]): EvalSummary {\n let passed = 0;\n let failed = 0;\n let skipped = 0;\n for (const r of results) {\n if (r.skipped) skipped++;\n else if (r.passed) passed++;\n else failed++;\n }\n return {\n total: results.length,\n passed,\n failed,\n skipped,\n allPassed: failed === 0,\n };\n}\n\n/** Status glyph for a single eval result. */\nexport function evalGlyph(result: EvalResult): string {\n if (result.skipped) return \"−\";\n return result.passed ? \"✓\" : \"✗\";\n}\n\n/** The one-line header for a single eval result (no failure detail). */\nexport function formatEvalHeadline(result: EvalResult): string {\n if (result.skipped) {\n return `− ${result.id} (skipped${\n result.skipped.reason ? `: ${result.skipped.reason}` : \"\"\n })`;\n }\n return `${evalGlyph(result)} ${result.id}${\n result.description ? ` — ${result.description}` : \"\"\n }`;\n}\n\n/** Indented detail lines for a failing eval (error + failing assertions). */\nexport function formatEvalDetail(result: EvalResult): string[] {\n const lines: string[] = [];\n if (result.error) lines.push(` error: ${result.error}`);\n for (const a of result.assertions) {\n if (a.pass) continue;\n const tag = a.severity === \"soft\" ? \"soft\" : \"gate\";\n lines.push(` ✗ [${tag}] ${a.label}${a.detail ? ` — ${a.detail}` : \"\"}`);\n }\n return lines;\n}\n\n/** The final PASS/FAIL summary line. */\nexport function formatSummaryLine(results: EvalResult[]): string {\n const s = summarize(results);\n return `${s.allPassed ? \"PASS\" : \"FAIL\"} — ${s.passed} passed, ${s.failed} failed, ${s.skipped} skipped (${s.total} total)`;\n}\n\n/** Render all results as a human-readable console report (non-streaming). */\nexport function formatEvalResults(results: EvalResult[]): string {\n const lines: string[] = [];\n for (const r of results) {\n lines.push(formatEvalHeadline(r));\n lines.push(...formatEvalDetail(r));\n }\n lines.push(\"\");\n lines.push(formatSummaryLine(results));\n return lines.join(\"\\n\");\n}\n"],"mappings":";AAWA,SAAgB,UAAU,SAAoC;CAC5D,IAAI,SAAS;CACb,IAAI,SAAS;CACb,IAAI,UAAU;AACd,MAAK,MAAM,KAAK,QACd,KAAI,EAAE,QAAS;UACN,EAAE,OAAQ;KACd;AAEP,QAAO;EACL,OAAO,QAAQ;EACf;EACA;EACA;EACA,WAAW,WAAW;EACvB;;;AAIH,SAAgB,UAAU,QAA4B;AACpD,KAAI,OAAO,QAAS,QAAO;AAC3B,QAAO,OAAO,SAAS,MAAM;;;AAI/B,SAAgB,mBAAmB,QAA4B;AAC7D,KAAI,OAAO,QACT,QAAO,KAAK,OAAO,GAAG,WACpB,OAAO,QAAQ,SAAS,KAAK,OAAO,QAAQ,WAAW,GACxD;AAEH,QAAO,GAAG,UAAU,OAAO,CAAC,GAAG,OAAO,KACpC,OAAO,cAAc,MAAM,OAAO,gBAAgB;;;AAKtD,SAAgB,iBAAiB,QAA8B;CAC7D,MAAM,QAAkB,EAAE;AAC1B,KAAI,OAAO,MAAO,OAAM,KAAK,cAAc,OAAO,QAAQ;AAC1D,MAAK,MAAM,KAAK,OAAO,YAAY;AACjC,MAAI,EAAE,KAAM;EACZ,MAAM,MAAM,EAAE,aAAa,SAAS,SAAS;AAC7C,QAAM,KAAK,UAAU,IAAI,IAAI,EAAE,QAAQ,EAAE,SAAS,MAAM,EAAE,WAAW,KAAK;;AAE5E,QAAO;;;AAIT,SAAgB,kBAAkB,SAA+B;CAC/D,MAAM,IAAI,UAAU,QAAQ;AAC5B,QAAO,GAAG,EAAE,YAAY,SAAS,OAAO,KAAK,EAAE,OAAO,WAAW,EAAE,OAAO,WAAW,EAAE,QAAQ,YAAY,EAAE,MAAM;;;AAIrH,SAAgB,kBAAkB,SAA+B;CAC/D,MAAM,QAAkB,EAAE;AAC1B,MAAK,MAAM,KAAK,SAAS;AACvB,QAAM,KAAK,mBAAmB,EAAE,CAAC;AACjC,QAAM,KAAK,GAAG,iBAAiB,EAAE,CAAC;;AAEpC,OAAM,KAAK,GAAG;AACd,OAAM,KAAK,kBAAkB,QAAQ,CAAC;AACtC,QAAO,MAAM,KAAK,KAAK"}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import { DatasetRow } from "./dataset.js";
|
|
2
|
+
import { EvalDefinition, EvalDriver, EvalResult } from "./types.js";
|
|
3
|
+
|
|
4
|
+
//#region src/evals/run-eval.d.ts
|
|
5
|
+
interface RunEvalOptions {
|
|
6
|
+
/** Stable id for the eval (e.g. its file path relative to the evals dir). */
|
|
7
|
+
id: string;
|
|
8
|
+
/** Drives the agent and returns reply/tool-calls/success per `send`. */
|
|
9
|
+
driver: EvalDriver;
|
|
10
|
+
/** When true, soft assertion failures also fail the eval. */
|
|
11
|
+
strict?: boolean;
|
|
12
|
+
/** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */
|
|
13
|
+
row?: DatasetRow;
|
|
14
|
+
}
|
|
15
|
+
/**
|
|
16
|
+
* Run a single eval against a driver. Never throws for assertion or agent
|
|
17
|
+
* failures — those become a non-passing {@link EvalResult}. Only a malformed
|
|
18
|
+
* eval definition surfaces as `result.error`.
|
|
19
|
+
*/
|
|
20
|
+
declare function runEval(def: EvalDefinition, options: RunEvalOptions): Promise<EvalResult>;
|
|
21
|
+
//#endregion
|
|
22
|
+
export { RunEvalOptions, runEval };
|
|
23
|
+
//# sourceMappingURL=run-eval.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-eval.d.ts","names":[],"sources":["../../src/evals/run-eval.ts"],"mappings":";;;;UAuBiB,cAAA;;EAEf,EAAA;EAF6B;EAI7B,MAAA,EAAQ,UAAA;EAIQ;EAFhB,MAAA;EAFA;EAIA,GAAA,GAAM,UAAA;AAAA;;;;;AAQR;iBAAsB,OAAA,CACpB,GAAA,EAAK,cAAA,EACL,OAAA,EAAS,cAAA,GACR,OAAA,CAAQ,UAAA"}
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import { judgeClosedQA, judgeCustom, judgeFactuality } from "./judge.js";
|
|
2
|
+
|
|
3
|
+
//#region src/evals/run-eval.ts
|
|
4
|
+
/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */
|
|
5
|
+
const DEFAULT_JUDGE_THRESHOLD = .5;
|
|
6
|
+
/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */
|
|
7
|
+
var SkipSignal = class extends Error {
|
|
8
|
+
constructor(reason) {
|
|
9
|
+
super("eval skipped");
|
|
10
|
+
this.reason = reason;
|
|
11
|
+
this.name = "SkipSignal";
|
|
12
|
+
}
|
|
13
|
+
};
|
|
14
|
+
/**
|
|
15
|
+
* Run a single eval against a driver. Never throws for assertion or agent
|
|
16
|
+
* failures — those become a non-passing {@link EvalResult}. Only a malformed
|
|
17
|
+
* eval definition surfaces as `result.error`.
|
|
18
|
+
*/
|
|
19
|
+
async function runEval(def, options) {
|
|
20
|
+
const assertions = [];
|
|
21
|
+
let reply = "";
|
|
22
|
+
let lastInput = "";
|
|
23
|
+
let toolCalls = [];
|
|
24
|
+
let sessionId;
|
|
25
|
+
let lastTraceId;
|
|
26
|
+
let lastSucceeded = false;
|
|
27
|
+
const record = (label, pass, score, detail) => {
|
|
28
|
+
const result = {
|
|
29
|
+
label,
|
|
30
|
+
severity: "gate",
|
|
31
|
+
pass,
|
|
32
|
+
score,
|
|
33
|
+
detail
|
|
34
|
+
};
|
|
35
|
+
assertions.push(result);
|
|
36
|
+
const handle = {
|
|
37
|
+
gate() {
|
|
38
|
+
result.severity = "gate";
|
|
39
|
+
return handle;
|
|
40
|
+
},
|
|
41
|
+
soft() {
|
|
42
|
+
result.severity = "soft";
|
|
43
|
+
return handle;
|
|
44
|
+
},
|
|
45
|
+
atLeast(threshold) {
|
|
46
|
+
result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;
|
|
47
|
+
return handle;
|
|
48
|
+
}
|
|
49
|
+
};
|
|
50
|
+
return handle;
|
|
51
|
+
};
|
|
52
|
+
const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);
|
|
53
|
+
const t = {
|
|
54
|
+
async send(message) {
|
|
55
|
+
lastInput = message;
|
|
56
|
+
const r = await options.driver.send(message);
|
|
57
|
+
reply = r.reply;
|
|
58
|
+
toolCalls = r.toolCalls;
|
|
59
|
+
sessionId = r.sessionId;
|
|
60
|
+
lastSucceeded = r.succeeded;
|
|
61
|
+
if (r.traceId) lastTraceId = r.traceId;
|
|
62
|
+
},
|
|
63
|
+
reset() {
|
|
64
|
+
options.driver.reset?.();
|
|
65
|
+
},
|
|
66
|
+
get reply() {
|
|
67
|
+
return reply;
|
|
68
|
+
},
|
|
69
|
+
get toolCalls() {
|
|
70
|
+
return toolCalls;
|
|
71
|
+
},
|
|
72
|
+
get sessionId() {
|
|
73
|
+
return sessionId;
|
|
74
|
+
},
|
|
75
|
+
get input() {
|
|
76
|
+
return options.row?.inputs ?? {};
|
|
77
|
+
},
|
|
78
|
+
get expected() {
|
|
79
|
+
return options.row?.expectations;
|
|
80
|
+
},
|
|
81
|
+
succeeded() {
|
|
82
|
+
return record("succeeded", lastSucceeded, void 0, lastSucceeded ? void 0 : "agent turn did not complete successfully");
|
|
83
|
+
},
|
|
84
|
+
calledTool(name) {
|
|
85
|
+
return record(`calledTool(${name})`, toolCalls.includes(name), void 0, `expected tool "${name}" to be called (called: ${toolCalls.length ? toolCalls.join(", ") : "none"})`);
|
|
86
|
+
},
|
|
87
|
+
check(value, matcher) {
|
|
88
|
+
const m = matcher(value);
|
|
89
|
+
return record("check", m.pass, m.score, m.detail);
|
|
90
|
+
},
|
|
91
|
+
judge: {
|
|
92
|
+
async factuality(expected) {
|
|
93
|
+
const { score, rationale } = await judgeFactuality({
|
|
94
|
+
input: lastInput,
|
|
95
|
+
output: reply,
|
|
96
|
+
expected
|
|
97
|
+
});
|
|
98
|
+
return recordJudge("judge.factuality", score, rationale);
|
|
99
|
+
},
|
|
100
|
+
async closedQA(criteria) {
|
|
101
|
+
const { score, rationale } = await judgeClosedQA({
|
|
102
|
+
input: lastInput,
|
|
103
|
+
output: reply,
|
|
104
|
+
criteria
|
|
105
|
+
});
|
|
106
|
+
return recordJudge("judge.closedQA", score, rationale);
|
|
107
|
+
},
|
|
108
|
+
async custom(spec) {
|
|
109
|
+
const { score, rationale } = await judgeCustom(spec, {
|
|
110
|
+
input: lastInput,
|
|
111
|
+
output: reply
|
|
112
|
+
});
|
|
113
|
+
return recordJudge(`judge.${spec.name}`, score, rationale);
|
|
114
|
+
}
|
|
115
|
+
},
|
|
116
|
+
skip(reason) {
|
|
117
|
+
throw new SkipSignal(reason);
|
|
118
|
+
}
|
|
119
|
+
};
|
|
120
|
+
try {
|
|
121
|
+
await def.test(t);
|
|
122
|
+
} catch (err) {
|
|
123
|
+
if (err instanceof SkipSignal) return {
|
|
124
|
+
id: options.id,
|
|
125
|
+
description: def.description,
|
|
126
|
+
skipped: { reason: err.reason },
|
|
127
|
+
assertions,
|
|
128
|
+
passed: true,
|
|
129
|
+
traceId: lastTraceId
|
|
130
|
+
};
|
|
131
|
+
return {
|
|
132
|
+
id: options.id,
|
|
133
|
+
description: def.description,
|
|
134
|
+
assertions,
|
|
135
|
+
passed: false,
|
|
136
|
+
error: err instanceof Error ? err.message : String(err),
|
|
137
|
+
traceId: lastTraceId
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
const passed = assertions.every((a) => a.pass || a.severity === "soft" && !options.strict);
|
|
141
|
+
return {
|
|
142
|
+
id: options.id,
|
|
143
|
+
description: def.description,
|
|
144
|
+
assertions,
|
|
145
|
+
passed,
|
|
146
|
+
traceId: lastTraceId
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
//#endregion
|
|
151
|
+
export { runEval };
|
|
152
|
+
//# sourceMappingURL=run-eval.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import type { DatasetRow } from \"./dataset\";\nimport { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */\n row?: DatasetRow;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n // Re-threshold the score; keep the current severity (gate unless the\n // caller also chained `.soft()`).\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and gate by default (a miss fails the eval,\n // like other assertions). The caller chains `.atLeast(n)` to change the pass\n // threshold, or `.soft()` to demote to a tracked-only metric.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message);\n reply = r.reply;\n toolCalls = r.toolCalls;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (r.traceId) lastTraceId = r.traceId;\n },\n reset() {\n options.driver.reset?.();\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n get input() {\n return options.row?.inputs ?? {};\n },\n get expected() {\n return options.row?.expectations;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n try {\n await def.test(t);\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n };\n}\n"],"mappings":";;;;AAaA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAoBhB,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AAGzB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAMT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU;CAEnE,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,QAAQ;AAC5C,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,QAAQ;AACN,WAAQ,OAAO,SAAS;;EAE1B,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,QAAQ;AACV,UAAO,QAAQ,KAAK,UAAU,EAAE;;EAElC,IAAI,WAAW;AACb,UAAO,QAAQ,KAAK;;EAEtB,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;AAED,KAAI;AACF,QAAM,IAAI,KAAK,EAAE;UACV,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;;CAGH,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EACV"}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { WorkspaceClient } from "../workspace-client/index.js";
|
|
2
|
+
import "./dataset.js";
|
|
3
|
+
import { EvalResult } from "./types.js";
|
|
4
|
+
import { ReportOutcome } from "./mlflow-report.js";
|
|
5
|
+
import { FinishOutcome } from "./mlflow-run.js";
|
|
6
|
+
|
|
7
|
+
//#region src/evals/run-evals.d.ts
|
|
8
|
+
interface RunEvalsOptions {
|
|
9
|
+
/** Project root containing `server/agents/`. Defaults to `process.cwd()`. */
|
|
10
|
+
rootDir?: string;
|
|
11
|
+
/** Base URL of the running app to drive, e.g. `http://localhost:3000`. */
|
|
12
|
+
baseUrl: string;
|
|
13
|
+
/** Substring filter on `<agent>/<id>` (or an exact agent id). */
|
|
14
|
+
filter?: string;
|
|
15
|
+
/** Soft assertion failures also fail the eval. */
|
|
16
|
+
strict?: boolean;
|
|
17
|
+
/** Extra request headers for the driver (e.g. auth for a deployed app). */
|
|
18
|
+
headers?: Record<string, string>;
|
|
19
|
+
/** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */
|
|
20
|
+
timeoutMs?: number;
|
|
21
|
+
/**
|
|
22
|
+
* Max evals to drive concurrently. Each eval opens one stream to the app as
|
|
23
|
+
* the same user, so keep this at or below the app's
|
|
24
|
+
* `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the
|
|
25
|
+
* 429 guard. Defaults to 4; clamped to `[1, total]`.
|
|
26
|
+
*/
|
|
27
|
+
concurrency?: number;
|
|
28
|
+
/**
|
|
29
|
+
* When set, create a native MLflow "Evaluation run": each eval's trace is
|
|
30
|
+
* linked to the run, pass/fail is written as feedback, and aggregate metrics
|
|
31
|
+
* are logged. Requires Databricks creds + the target experiment.
|
|
32
|
+
*/
|
|
33
|
+
mlflow?: {
|
|
34
|
+
host: string;
|
|
35
|
+
token: string;
|
|
36
|
+
experimentId: string; /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */
|
|
37
|
+
sqlWarehouseId?: string;
|
|
38
|
+
};
|
|
39
|
+
/**
|
|
40
|
+
* When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a
|
|
41
|
+
* Databricks serving endpoint (`model`).
|
|
42
|
+
*/
|
|
43
|
+
judge?: {
|
|
44
|
+
host: string;
|
|
45
|
+
token: string;
|
|
46
|
+
model: string;
|
|
47
|
+
};
|
|
48
|
+
/**
|
|
49
|
+
* Workspace client used to read managed evaluation datasets (for evals that
|
|
50
|
+
* declare `dataset`). Required alongside {@link warehouseId} for those evals.
|
|
51
|
+
*/
|
|
52
|
+
workspaceClient?: WorkspaceClient;
|
|
53
|
+
/** SQL warehouse id used to read managed evaluation datasets. */
|
|
54
|
+
warehouseId?: string;
|
|
55
|
+
/** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */
|
|
56
|
+
now?: number;
|
|
57
|
+
/** Progress callback, invoked as evals are discovered, started, and finished. */
|
|
58
|
+
onEvent?: (event: EvalProgress) => void;
|
|
59
|
+
}
|
|
60
|
+
type EvalProgress = {
|
|
61
|
+
type: "discovered";
|
|
62
|
+
total: number;
|
|
63
|
+
} | {
|
|
64
|
+
type: "run-created";
|
|
65
|
+
runId: string;
|
|
66
|
+
} | {
|
|
67
|
+
type: "start";
|
|
68
|
+
id: string;
|
|
69
|
+
index: number;
|
|
70
|
+
total: number;
|
|
71
|
+
} | {
|
|
72
|
+
type: "result";
|
|
73
|
+
result: EvalResult;
|
|
74
|
+
index: number;
|
|
75
|
+
total: number;
|
|
76
|
+
};
|
|
77
|
+
interface EvalRunSummary {
|
|
78
|
+
results: EvalResult[];
|
|
79
|
+
/** Present when an MLflow evaluation run was created. */
|
|
80
|
+
mlflow?: {
|
|
81
|
+
runId: string;
|
|
82
|
+
report: ReportOutcome;
|
|
83
|
+
finish: FinishOutcome;
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
88
|
+
* the agents on a running app. Never throws for an individual eval — load/run
|
|
89
|
+
* failures become non-passing {@link EvalResult}s.
|
|
90
|
+
*/
|
|
91
|
+
declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
|
|
92
|
+
//#endregion
|
|
93
|
+
export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir };
|
|
94
|
+
//# sourceMappingURL=run-evals.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAciB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAMU;EAJV,MAAA;EAyCkB;EAvClB,MAAA;EAuC8B;EArC9B,OAAA,GAAU,MAAA;EANV;EAQA,SAAA;EAJA;;;;;;EAWA,WAAA;EAQE;;;;;EAFF,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UAeF;IAbE,cAAA;EAAA;EAiBgB;;;;EAXlB,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EAef;;;;EAVxB,eAAA,GAAkB,eAAA;EAYa;EAV/B,WAAA;EAWI;EATJ,GAAA;EAS4B;EAP5B,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EAE6D;EAAtE,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;iBAoOrC,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
|
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
|
+
import { readEvalDataset } from "./dataset.js";
|
|
3
|
+
import { discoverEvalFiles } from "./discover.js";
|
|
4
|
+
import { createHttpDriver } from "./http-driver.js";
|
|
5
|
+
import { configureJudge, teardownJudge } from "./judge.js";
|
|
6
|
+
import { mapPool } from "./pool.js";
|
|
7
|
+
import { reportToMlflow } from "./mlflow-report.js";
|
|
8
|
+
import { runEval } from "./run-eval.js";
|
|
9
|
+
import { createEvalRun, finishEvalRun } from "./mlflow-run.js";
|
|
10
|
+
import { pathToFileURL } from "node:url";
|
|
11
|
+
|
|
12
|
+
//#region src/evals/run-evals.ts
|
|
13
|
+
/**
|
|
14
|
+
* Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.
|
|
15
|
+
* Uses tsx's programmatic loader so TypeScript eval files run without a build
|
|
16
|
+
* step. The specifier is indirected so the type checker doesn't try to resolve
|
|
17
|
+
* tsx's internal entry.
|
|
18
|
+
*/
|
|
19
|
+
async function loadEval(file) {
|
|
20
|
+
const tsxApi = "tsx/esm/api";
|
|
21
|
+
let tsImport;
|
|
22
|
+
try {
|
|
23
|
+
({tsImport} = await import(tsxApi));
|
|
24
|
+
} catch {
|
|
25
|
+
throw new Error("Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).");
|
|
26
|
+
}
|
|
27
|
+
const def = resolveEvalDefault(await tsImport(pathToFileURL(file).href, import.meta.url));
|
|
28
|
+
if (!def) throw new Error(`${file}: must default-export defineEval({ test })`);
|
|
29
|
+
return def;
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Unwrap the eval default export across module-interop shapes. Depending on
|
|
33
|
+
* whether the eval file is treated as ESM or CJS, the value lands at
|
|
34
|
+
* `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or
|
|
35
|
+
* `mod` itself. Returns the first candidate that looks like an eval.
|
|
36
|
+
*/
|
|
37
|
+
function resolveEvalDefault(mod) {
|
|
38
|
+
let candidate = mod;
|
|
39
|
+
for (let i = 0; i < 4 && candidate; i++) {
|
|
40
|
+
if (typeof candidate.test === "function") return candidate;
|
|
41
|
+
candidate = candidate.default;
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Run one eval turn against a fresh driver. Never throws — a run failure becomes
|
|
46
|
+
* a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`
|
|
47
|
+
* binds the current managed-dataset row (see {@link resolveDatasetRows}), or is
|
|
48
|
+
* `undefined` for a plain single-run eval.
|
|
49
|
+
*/
|
|
50
|
+
async function runOne(d, id, def, row, runId, options) {
|
|
51
|
+
try {
|
|
52
|
+
return await runEval(def, {
|
|
53
|
+
id,
|
|
54
|
+
driver: createHttpDriver({
|
|
55
|
+
baseUrl: options.baseUrl,
|
|
56
|
+
agent: def.agent ?? d.agent,
|
|
57
|
+
headers: options.headers,
|
|
58
|
+
mlflowRunId: runId,
|
|
59
|
+
timeoutMs: options.timeoutMs
|
|
60
|
+
}),
|
|
61
|
+
strict: options.strict,
|
|
62
|
+
row
|
|
63
|
+
});
|
|
64
|
+
} catch (err) {
|
|
65
|
+
return {
|
|
66
|
+
id,
|
|
67
|
+
assertions: [],
|
|
68
|
+
passed: false,
|
|
69
|
+
error: err instanceof Error ? err.message : String(err)
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval
|
|
75
|
+
* yields a single `undefined` row; a dataset eval reads its Unity Catalog table
|
|
76
|
+
* via {@link readEvalDataset}. On misconfiguration or read failure, returns a
|
|
77
|
+
* single `undefined` row plus an `error`, so the eval still surfaces one result.
|
|
78
|
+
*/
|
|
79
|
+
async function resolveDatasetRows(def, options) {
|
|
80
|
+
if (!def.dataset) return { rows: [void 0] };
|
|
81
|
+
if (!options.workspaceClient || !options.warehouseId) return {
|
|
82
|
+
rows: [void 0],
|
|
83
|
+
error: "dataset eval requires a workspace client and warehouse (pass --warehouse-id)"
|
|
84
|
+
};
|
|
85
|
+
try {
|
|
86
|
+
const rows = await readEvalDataset(options.workspaceClient, {
|
|
87
|
+
table: def.dataset.table,
|
|
88
|
+
warehouseId: options.warehouseId,
|
|
89
|
+
limit: def.dataset.limit
|
|
90
|
+
});
|
|
91
|
+
if (rows.length === 0) return {
|
|
92
|
+
rows: [void 0],
|
|
93
|
+
error: `dataset "${def.dataset.table}" returned no rows`
|
|
94
|
+
};
|
|
95
|
+
return { rows };
|
|
96
|
+
} catch (err) {
|
|
97
|
+
return {
|
|
98
|
+
rows: [void 0],
|
|
99
|
+
error: err instanceof Error ? err.message : String(err)
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* Load one discovered eval and run it, expanding a dataset-driven eval into one
|
|
105
|
+
* run per row. Appends one result per row to `results`, emitting `start`/
|
|
106
|
+
* `result` around each. Never throws: a load or dataset-read failure surfaces as
|
|
107
|
+
* a non-passing result. `total` counts eval files, not rows — per-row detail is
|
|
108
|
+
* carried in the result id (`[row i/n]`).
|
|
109
|
+
*/
|
|
110
|
+
async function runDiscovered(d, index, total, runId, options, emit, results) {
|
|
111
|
+
const id = `${d.agent}/${d.id}`;
|
|
112
|
+
let def;
|
|
113
|
+
try {
|
|
114
|
+
def = await loadEval(d.file);
|
|
115
|
+
} catch (err) {
|
|
116
|
+
emit({
|
|
117
|
+
type: "start",
|
|
118
|
+
id,
|
|
119
|
+
index,
|
|
120
|
+
total
|
|
121
|
+
});
|
|
122
|
+
const result = {
|
|
123
|
+
id,
|
|
124
|
+
assertions: [],
|
|
125
|
+
passed: false,
|
|
126
|
+
error: err instanceof Error ? err.message : String(err)
|
|
127
|
+
};
|
|
128
|
+
results.push(result);
|
|
129
|
+
emit({
|
|
130
|
+
type: "result",
|
|
131
|
+
result,
|
|
132
|
+
index,
|
|
133
|
+
total
|
|
134
|
+
});
|
|
135
|
+
return;
|
|
136
|
+
}
|
|
137
|
+
const { rows, error: datasetError } = await resolveDatasetRows(def, options);
|
|
138
|
+
for (let r = 0; r < rows.length; r++) {
|
|
139
|
+
const rowId = def.dataset && rows.length > 1 ? `${id} [row ${r + 1}/${rows.length}]` : id;
|
|
140
|
+
emit({
|
|
141
|
+
type: "start",
|
|
142
|
+
id: rowId,
|
|
143
|
+
index,
|
|
144
|
+
total
|
|
145
|
+
});
|
|
146
|
+
const result = datasetError ? {
|
|
147
|
+
id: rowId,
|
|
148
|
+
assertions: [],
|
|
149
|
+
passed: false,
|
|
150
|
+
error: datasetError
|
|
151
|
+
} : await runOne(d, rowId, def, rows[r], runId, options);
|
|
152
|
+
results.push(result);
|
|
153
|
+
emit({
|
|
154
|
+
type: "result",
|
|
155
|
+
result,
|
|
156
|
+
index,
|
|
157
|
+
total
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */
|
|
162
|
+
async function maybeConfigureJudge(options) {
|
|
163
|
+
if (!options.judge) return;
|
|
164
|
+
await configureJudge({
|
|
165
|
+
client: new MlflowClient(options.judge.host, options.judge.token),
|
|
166
|
+
token: options.judge.token,
|
|
167
|
+
model: options.judge.model
|
|
168
|
+
});
|
|
169
|
+
}
|
|
170
|
+
/**
|
|
171
|
+
* Report per-eval assessments and finish the MLflow run, when one was created.
|
|
172
|
+
* Returns the run summary, or `undefined` when there was no run to finalize.
|
|
173
|
+
*/
|
|
174
|
+
async function finalizeMlflow(client, runId, results, options) {
|
|
175
|
+
if (!client || !runId) return void 0;
|
|
176
|
+
let report = {
|
|
177
|
+
written: 0,
|
|
178
|
+
skipped: 0,
|
|
179
|
+
failures: []
|
|
180
|
+
};
|
|
181
|
+
try {
|
|
182
|
+
report = await reportToMlflow(client, results, options.mlflow?.sqlWarehouseId);
|
|
183
|
+
} catch (err) {
|
|
184
|
+
report.failures.push({
|
|
185
|
+
traceId: "(report)",
|
|
186
|
+
error: err instanceof Error ? err.message : String(err)
|
|
187
|
+
});
|
|
188
|
+
}
|
|
189
|
+
const finish = await finishEvalRun(client, {
|
|
190
|
+
runId,
|
|
191
|
+
results,
|
|
192
|
+
endTime: options.now ?? Date.now()
|
|
193
|
+
});
|
|
194
|
+
return {
|
|
195
|
+
runId,
|
|
196
|
+
report,
|
|
197
|
+
finish
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
/**
|
|
201
|
+
* Default max evals in flight. Each eval opens one stream to the app as the
|
|
202
|
+
* same user; the server caps concurrent streams per user at 5 by default
|
|
203
|
+
* (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.
|
|
204
|
+
*/
|
|
205
|
+
const DEFAULT_CONCURRENCY = 4;
|
|
206
|
+
/**
|
|
207
|
+
* Discover, load, and run every eval under each agent's `evals/` dir, driving
|
|
208
|
+
* the agents on a running app. Never throws for an individual eval — load/run
|
|
209
|
+
* failures become non-passing {@link EvalResult}s.
|
|
210
|
+
*/
|
|
211
|
+
async function runEvalsInDir(options) {
|
|
212
|
+
const root = options.rootDir ?? process.cwd();
|
|
213
|
+
const now = options.now ?? Date.now();
|
|
214
|
+
let discovered = discoverEvalFiles(root);
|
|
215
|
+
if (options.filter) {
|
|
216
|
+
const f = options.filter;
|
|
217
|
+
discovered = discovered.filter((d) => d.agent === f || `${d.agent}/${d.id}`.includes(f));
|
|
218
|
+
}
|
|
219
|
+
const emit = options.onEvent ?? (() => {});
|
|
220
|
+
const total = discovered.length;
|
|
221
|
+
emit({
|
|
222
|
+
type: "discovered",
|
|
223
|
+
total
|
|
224
|
+
});
|
|
225
|
+
await maybeConfigureJudge(options);
|
|
226
|
+
try {
|
|
227
|
+
let runId;
|
|
228
|
+
let mlflowClient;
|
|
229
|
+
if (options.mlflow) {
|
|
230
|
+
mlflowClient = new MlflowClient(options.mlflow.host, options.mlflow.token);
|
|
231
|
+
runId = await createEvalRun(mlflowClient, {
|
|
232
|
+
experimentId: options.mlflow.experimentId,
|
|
233
|
+
runName: `appkit-eval ${new Date(now).toISOString()}`,
|
|
234
|
+
startTime: now
|
|
235
|
+
});
|
|
236
|
+
emit({
|
|
237
|
+
type: "run-created",
|
|
238
|
+
runId
|
|
239
|
+
});
|
|
240
|
+
}
|
|
241
|
+
const results = (await mapPool(discovered, options.concurrency ?? DEFAULT_CONCURRENCY, async (d, index) => {
|
|
242
|
+
const fileResults = [];
|
|
243
|
+
await runDiscovered(d, index, total, runId, options, emit, fileResults);
|
|
244
|
+
return fileResults;
|
|
245
|
+
})).flat();
|
|
246
|
+
const summary = { results };
|
|
247
|
+
const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);
|
|
248
|
+
if (mlflow) summary.mlflow = mlflow;
|
|
249
|
+
return summary;
|
|
250
|
+
} finally {
|
|
251
|
+
teardownJudge();
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
//#endregion
|
|
256
|
+
export { runEvalsInDir };
|
|
257
|
+
//# sourceMappingURL=run-evals.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-evals.js","names":[],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport { type DiscoveredEval, discoverEvalFiles } from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */\n timeoutMs?: number;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n * Uses tsx's programmatic loader so TypeScript eval files run without a build\n * step. The specifier is indirected so the type checker doesn't try to resolve\n * tsx's internal entry.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n\n const mod = await tsImport(pathToFileURL(file).href, import.meta.url);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // A fresh driver per row: each row is an independent conversation whose\n // thread must not carry over the previous row's history. (Multiple\n // `t.send`s within one row still share the thread — the driver's behavior.)\n const driver = createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n timeoutMs: options.timeoutMs,\n });\n return await runEval(def, { id, driver, strict: options.strict, row });\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Load one discovered eval and run it, expanding a dataset-driven eval into one\n * run per row. Appends one result per row to `results`, emitting `start`/\n * `result` around each. Never throws: a load or dataset-read failure surfaces as\n * a non-passing result. `total` counts eval files, not rows — per-row detail is\n * carried in the result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n const total = discovered.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each eval through the bounded pool — one in-flight stream per eval, so\n // the pool respects the server's per-user stream cap (see mapPool/concurrency).\n // A dataset eval expands into per-row runs that execute serially within its\n // slot; results preserve discovery order (mapPool writes by index) and row\n // order within each file. `total` counts eval files, not dataset rows — per-row\n // detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n discovered,\n options.concurrency ?? DEFAULT_CONCURRENCY,\n async (d, index) => {\n const fileResults: EvalResult[] = [];\n await runDiscovered(d, index, total, runId, options, emit, fileResults);\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAkFA,eAAe,SAAS,MAAuC;CAC7D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;CAIH,MAAM,MAAM,mBADA,MAAM,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI,CAClC;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;;AAST,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAWF,SAAO,MAAM,QAAQ,KAAK;GAAE;GAAI,QAPjB,iBAAiB;IAC9B,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IACb,WAAW,QAAQ;IACpB,CAAC;GACsC,QAAQ,QAAQ;GAAQ;GAAK,CAAC;UAC/D,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;AAWL,eAAe,cACb,GACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;CAE3B,IAAI;AACJ,KAAI;AACF,QAAM,MAAM,SAAS,EAAE,KAAK;UACrB,KAAK;AACZ,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;AAO5B,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CACvC,MAAM,QAAQ,WAAW;AACzB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkBtC,MAAM,WATU,MAAM,QACpB,YACA,QAAQ,eAAe,qBACvB,OAAO,GAAG,UAAU;GAClB,MAAM,cAA4B,EAAE;AACpC,SAAM,cAAc,GAAG,OAAO,OAAO,OAAO,SAAS,MAAM,YAAY;AACvE,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
|