@databricks/appkit 0.72.0 → 0.73.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +17 -0
- package/NOTICE.md +1 -0
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +7 -4
- package/dist/beta.js +3 -2
- package/dist/cli/commands/agent/eval.js +9 -1
- package/dist/cli/commands/agent/eval.js.map +1 -1
- package/dist/connectors/index.js +1 -1
- package/dist/connectors/mlflow/auth.d.ts +11 -1
- package/dist/connectors/mlflow/auth.d.ts.map +1 -1
- package/dist/connectors/mlflow/auth.js +22 -2
- package/dist/connectors/mlflow/auth.js.map +1 -1
- package/dist/connectors/mlflow/index.d.ts +2 -0
- package/dist/database/errors.js +15 -5
- package/dist/database/errors.js.map +1 -1
- package/dist/database/runtime/data-path.d.ts +7 -0
- package/dist/database/runtime/data-path.d.ts.map +1 -0
- package/dist/database/runtime/data-path.js.map +1 -1
- package/dist/database/runtime/engine/drizzle-data-path.js +7 -5
- package/dist/database/runtime/engine/drizzle-data-path.js.map +1 -1
- package/dist/database/schema-builder/define-schema.d.ts +1 -1
- package/dist/database/schema-builder/define-schema.js +1 -1
- package/dist/database/schema-builder/define-schema.js.map +1 -1
- package/dist/errors/database-validation.d.ts +23 -0
- package/dist/errors/database-validation.d.ts.map +1 -0
- package/dist/errors/database-validation.js +24 -0
- package/dist/errors/database-validation.js.map +1 -0
- package/dist/errors/index.js +1 -0
- package/dist/evals/dataset.d.ts +36 -0
- package/dist/evals/dataset.d.ts.map +1 -0
- package/dist/evals/dataset.js +36 -0
- package/dist/evals/dataset.js.map +1 -0
- package/dist/evals/http-driver.js +56 -51
- package/dist/evals/http-driver.js.map +1 -1
- package/dist/evals/index.d.ts +14 -0
- package/dist/evals/index.js +2 -1
- package/dist/evals/judge.d.ts +1 -0
- package/dist/evals/judge.d.ts.map +1 -1
- package/dist/evals/mlflow-report.d.ts +1 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -1
- package/dist/evals/mlflow-run.d.ts +2 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -1
- package/dist/evals/run-eval.d.ts +3 -0
- package/dist/evals/run-eval.d.ts.map +1 -1
- package/dist/evals/run-eval.js +10 -2
- package/dist/evals/run-eval.js.map +1 -1
- package/dist/evals/run-evals.d.ts +9 -0
- package/dist/evals/run-evals.d.ts.map +1 -1
- package/dist/evals/run-evals.js +101 -22
- package/dist/evals/run-evals.js.map +1 -1
- package/dist/evals/types.d.ts +40 -4
- package/dist/evals/types.d.ts.map +1 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/plugin/plugin.d.ts.map +1 -1
- package/dist/plugin/plugin.js +1 -1
- package/dist/plugin/plugin.js.map +1 -1
- package/dist/plugins/database/crud/contract.js +17 -8
- package/dist/plugins/database/crud/contract.js.map +1 -1
- package/dist/plugins/database/crud/exposure.js +63 -22
- package/dist/plugins/database/crud/exposure.js.map +1 -1
- package/dist/plugins/database/crud/request.js +50 -0
- package/dist/plugins/database/crud/request.js.map +1 -0
- package/dist/plugins/database/crud/response.js +77 -0
- package/dist/plugins/database/crud/response.js.map +1 -0
- package/dist/plugins/database/crud/routes.js +71 -52
- package/dist/plugins/database/crud/routes.js.map +1 -1
- package/dist/plugins/database/database.d.ts +6 -4
- package/dist/plugins/database/database.d.ts.map +1 -1
- package/dist/plugins/database/database.js +46 -16
- package/dist/plugins/database/database.js.map +1 -1
- package/dist/plugins/database/defaults.js +5 -1
- package/dist/plugins/database/defaults.js.map +1 -1
- package/dist/plugins/database/entity-client.js +143 -10
- package/dist/plugins/database/entity-client.js.map +1 -1
- package/dist/plugins/database/entity-types.d.ts +1 -1
- package/dist/plugins/database/hooks.d.ts +38 -0
- package/dist/plugins/database/hooks.d.ts.map +1 -0
- package/dist/plugins/database/index.d.ts +3 -2
- package/dist/plugins/database/lifecycle.js +67 -28
- package/dist/plugins/database/lifecycle.js.map +1 -1
- package/dist/plugins/database/scope.js +58 -0
- package/dist/plugins/database/scope.js.map +1 -0
- package/dist/plugins/database/types.d.ts +40 -12
- package/dist/plugins/database/types.d.ts.map +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +33 -33
- package/docs/api/appkit/Class.AppKitError.md +1 -0
- package/docs/api/appkit/Class.DatabaseValidationError.md +191 -0
- package/docs/api/appkit/Function.defineSchema.md +1 -1
- package/docs/api/appkit/Function.readEvalDataset.md +21 -0
- package/docs/api/appkit/Function.resolveWorkspaceClient.md +18 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +1 -1
- package/docs/api/appkit/Interface.DatabaseValidationIssue.md +21 -0
- package/docs/api/appkit/Interface.DatasetRow.md +21 -0
- package/docs/api/appkit/Interface.EntityMutationHooks.md +173 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +28 -0
- package/docs/api/appkit/Interface.EvalDriver.md +16 -1
- package/docs/api/appkit/Interface.HookApp.md +12 -0
- package/docs/api/appkit/Interface.HookContext.md +21 -0
- package/docs/api/appkit/Interface.ReadEvalDatasetOptions.md +34 -0
- package/docs/api/appkit/Interface.ReadSerializerContext.md +21 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +11 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +22 -0
- package/docs/api/appkit/Interface.TestContext.md +41 -4
- package/docs/api/appkit/TypeAlias.DatabaseApiConfig.md +53 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md +8 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md +49 -0
- package/docs/api/appkit/TypeAlias.DatabaseExports.md +3 -3
- package/docs/api/appkit/TypeAlias.EntityHooks.md +25 -0
- package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +16 -5
- package/docs/api/appkit/TypeAlias.ReadSerializer.md +19 -0
- package/docs/api/appkit/TypeAlias.TransactionClient.md +19 -0
- package/docs/api/appkit.md +135 -119
- package/docs/plugins/database.md +144 -0
- package/llms.txt +17 -0
- package/package.json +2 -2
- package/sbom.cdx.json +1 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-eval.d.ts","names":[],"sources":["../../src/evals/run-eval.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"run-eval.d.ts","names":[],"sources":["../../src/evals/run-eval.ts"],"mappings":";;;;UAuBiB,cAAA;;EAEf,EAAA;EAF6B;EAI7B,MAAA,EAAQ,UAAA;EAIQ;EAFhB,MAAA;EAFA;EAIA,GAAA,GAAM,UAAA;AAAA;;;;;AAQR;iBAAsB,OAAA,CACpB,GAAA,EAAK,cAAA,EACL,OAAA,EAAS,cAAA,GACR,OAAA,CAAQ,UAAA"}
|
package/dist/evals/run-eval.js
CHANGED
|
@@ -43,14 +43,13 @@ async function runEval(def, options) {
|
|
|
43
43
|
return handle;
|
|
44
44
|
},
|
|
45
45
|
atLeast(threshold) {
|
|
46
|
-
result.severity = "soft";
|
|
47
46
|
result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;
|
|
48
47
|
return handle;
|
|
49
48
|
}
|
|
50
49
|
};
|
|
51
50
|
return handle;
|
|
52
51
|
};
|
|
53
|
-
const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale)
|
|
52
|
+
const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);
|
|
54
53
|
const t = {
|
|
55
54
|
async send(message) {
|
|
56
55
|
lastInput = message;
|
|
@@ -61,6 +60,9 @@ async function runEval(def, options) {
|
|
|
61
60
|
lastSucceeded = r.succeeded;
|
|
62
61
|
if (r.traceId) lastTraceId = r.traceId;
|
|
63
62
|
},
|
|
63
|
+
reset() {
|
|
64
|
+
options.driver.reset?.();
|
|
65
|
+
},
|
|
64
66
|
get reply() {
|
|
65
67
|
return reply;
|
|
66
68
|
},
|
|
@@ -70,6 +72,12 @@ async function runEval(def, options) {
|
|
|
70
72
|
get sessionId() {
|
|
71
73
|
return sessionId;
|
|
72
74
|
},
|
|
75
|
+
get input() {
|
|
76
|
+
return options.row?.inputs ?? {};
|
|
77
|
+
},
|
|
78
|
+
get expected() {
|
|
79
|
+
return options.row?.expectations;
|
|
80
|
+
},
|
|
73
81
|
succeeded() {
|
|
74
82
|
return record("succeeded", lastSucceeded, void 0, lastSucceeded ? void 0 : "agent turn did not complete successfully");
|
|
75
83
|
},
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n
|
|
1
|
+
{"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import type { DatasetRow } from \"./dataset\";\nimport { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */\n row?: DatasetRow;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n // Re-threshold the score; keep the current severity (gate unless the\n // caller also chained `.soft()`).\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and gate by default (a miss fails the eval,\n // like other assertions). The caller chains `.atLeast(n)` to change the pass\n // threshold, or `.soft()` to demote to a tracked-only metric.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message);\n reply = r.reply;\n toolCalls = r.toolCalls;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (r.traceId) lastTraceId = r.traceId;\n },\n reset() {\n options.driver.reset?.();\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n get input() {\n return options.row?.inputs ?? {};\n },\n get expected() {\n return options.row?.expectations;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n try {\n await def.test(t);\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n };\n}\n"],"mappings":";;;;AAaA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAoBhB,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AAGzB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAMT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU;CAEnE,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,QAAQ;AAC5C,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,QAAQ;AACN,WAAQ,OAAO,SAAS;;EAE1B,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,QAAQ;AACV,UAAO,QAAQ,KAAK,UAAU,EAAE;;EAElC,IAAI,WAAW;AACb,UAAO,QAAQ,KAAK;;EAEtB,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;AAED,KAAI;AACF,QAAM,IAAI,KAAK,EAAE;UACV,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;;CAGH,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EACV"}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { WorkspaceClient } from "../workspace-client/index.js";
|
|
2
|
+
import "./dataset.js";
|
|
1
3
|
import { EvalResult } from "./types.js";
|
|
2
4
|
import { ReportOutcome } from "./mlflow-report.js";
|
|
3
5
|
import { FinishOutcome } from "./mlflow-run.js";
|
|
@@ -43,6 +45,13 @@ interface RunEvalsOptions {
|
|
|
43
45
|
token: string;
|
|
44
46
|
model: string;
|
|
45
47
|
};
|
|
48
|
+
/**
|
|
49
|
+
* Workspace client used to read managed evaluation datasets (for evals that
|
|
50
|
+
* declare `dataset`). Required alongside {@link warehouseId} for those evals.
|
|
51
|
+
*/
|
|
52
|
+
workspaceClient?: WorkspaceClient;
|
|
53
|
+
/** SQL warehouse id used to read managed evaluation datasets. */
|
|
54
|
+
warehouseId?: string;
|
|
46
55
|
/** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */
|
|
47
56
|
now?: number;
|
|
48
57
|
/** Progress callback, invoked as evals are discovered, started, and finished. */
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAciB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAMU;EAJV,MAAA;EAyCkB;EAvClB,MAAA;EAuC8B;EArC9B,OAAA,GAAU,MAAA;EANV;EAQA,SAAA;EAJA;;;;;;EAWA,WAAA;EAQE;;;;;EAFF,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UAeF;IAbE,cAAA;EAAA;EAiBgB;;;;EAXlB,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EAef;;;;EAVxB,eAAA,GAAkB,eAAA;EAYa;EAV/B,WAAA;EAWI;EATJ,GAAA;EAS4B;EAP5B,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EAE6D;EAAtE,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;iBAoOrC,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
|
package/dist/evals/run-evals.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { MlflowClient } from "../connectors/mlflow/client.js";
|
|
2
|
+
import { readEvalDataset } from "./dataset.js";
|
|
2
3
|
import { discoverEvalFiles } from "./discover.js";
|
|
3
4
|
import { createHttpDriver } from "./http-driver.js";
|
|
4
5
|
import { configureJudge, teardownJudge } from "./judge.js";
|
|
@@ -41,12 +42,13 @@ function resolveEvalDefault(mod) {
|
|
|
41
42
|
}
|
|
42
43
|
}
|
|
43
44
|
/**
|
|
44
|
-
*
|
|
45
|
-
*
|
|
45
|
+
* Run one eval turn against a fresh driver. Never throws — a run failure becomes
|
|
46
|
+
* a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`
|
|
47
|
+
* binds the current managed-dataset row (see {@link resolveDatasetRows}), or is
|
|
48
|
+
* `undefined` for a plain single-run eval.
|
|
46
49
|
*/
|
|
47
|
-
async function runOne(d, id, runId, options) {
|
|
50
|
+
async function runOne(d, id, def, row, runId, options) {
|
|
48
51
|
try {
|
|
49
|
-
const def = await loadEval(d.file);
|
|
50
52
|
return await runEval(def, {
|
|
51
53
|
id,
|
|
52
54
|
driver: createHttpDriver({
|
|
@@ -56,7 +58,8 @@ async function runOne(d, id, runId, options) {
|
|
|
56
58
|
mlflowRunId: runId,
|
|
57
59
|
timeoutMs: options.timeoutMs
|
|
58
60
|
}),
|
|
59
|
-
strict: options.strict
|
|
61
|
+
strict: options.strict,
|
|
62
|
+
row
|
|
60
63
|
});
|
|
61
64
|
} catch (err) {
|
|
62
65
|
return {
|
|
@@ -67,6 +70,94 @@ async function runOne(d, id, runId, options) {
|
|
|
67
70
|
};
|
|
68
71
|
}
|
|
69
72
|
}
|
|
73
|
+
/**
|
|
74
|
+
* Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval
|
|
75
|
+
* yields a single `undefined` row; a dataset eval reads its Unity Catalog table
|
|
76
|
+
* via {@link readEvalDataset}. On misconfiguration or read failure, returns a
|
|
77
|
+
* single `undefined` row plus an `error`, so the eval still surfaces one result.
|
|
78
|
+
*/
|
|
79
|
+
async function resolveDatasetRows(def, options) {
|
|
80
|
+
if (!def.dataset) return { rows: [void 0] };
|
|
81
|
+
if (!options.workspaceClient || !options.warehouseId) return {
|
|
82
|
+
rows: [void 0],
|
|
83
|
+
error: "dataset eval requires a workspace client and warehouse (pass --warehouse-id)"
|
|
84
|
+
};
|
|
85
|
+
try {
|
|
86
|
+
const rows = await readEvalDataset(options.workspaceClient, {
|
|
87
|
+
table: def.dataset.table,
|
|
88
|
+
warehouseId: options.warehouseId,
|
|
89
|
+
limit: def.dataset.limit
|
|
90
|
+
});
|
|
91
|
+
if (rows.length === 0) return {
|
|
92
|
+
rows: [void 0],
|
|
93
|
+
error: `dataset "${def.dataset.table}" returned no rows`
|
|
94
|
+
};
|
|
95
|
+
return { rows };
|
|
96
|
+
} catch (err) {
|
|
97
|
+
return {
|
|
98
|
+
rows: [void 0],
|
|
99
|
+
error: err instanceof Error ? err.message : String(err)
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* Load one discovered eval and run it, expanding a dataset-driven eval into one
|
|
105
|
+
* run per row. Appends one result per row to `results`, emitting `start`/
|
|
106
|
+
* `result` around each. Never throws: a load or dataset-read failure surfaces as
|
|
107
|
+
* a non-passing result. `total` counts eval files, not rows — per-row detail is
|
|
108
|
+
* carried in the result id (`[row i/n]`).
|
|
109
|
+
*/
|
|
110
|
+
async function runDiscovered(d, index, total, runId, options, emit, results) {
|
|
111
|
+
const id = `${d.agent}/${d.id}`;
|
|
112
|
+
let def;
|
|
113
|
+
try {
|
|
114
|
+
def = await loadEval(d.file);
|
|
115
|
+
} catch (err) {
|
|
116
|
+
emit({
|
|
117
|
+
type: "start",
|
|
118
|
+
id,
|
|
119
|
+
index,
|
|
120
|
+
total
|
|
121
|
+
});
|
|
122
|
+
const result = {
|
|
123
|
+
id,
|
|
124
|
+
assertions: [],
|
|
125
|
+
passed: false,
|
|
126
|
+
error: err instanceof Error ? err.message : String(err)
|
|
127
|
+
};
|
|
128
|
+
results.push(result);
|
|
129
|
+
emit({
|
|
130
|
+
type: "result",
|
|
131
|
+
result,
|
|
132
|
+
index,
|
|
133
|
+
total
|
|
134
|
+
});
|
|
135
|
+
return;
|
|
136
|
+
}
|
|
137
|
+
const { rows, error: datasetError } = await resolveDatasetRows(def, options);
|
|
138
|
+
for (let r = 0; r < rows.length; r++) {
|
|
139
|
+
const rowId = def.dataset && rows.length > 1 ? `${id} [row ${r + 1}/${rows.length}]` : id;
|
|
140
|
+
emit({
|
|
141
|
+
type: "start",
|
|
142
|
+
id: rowId,
|
|
143
|
+
index,
|
|
144
|
+
total
|
|
145
|
+
});
|
|
146
|
+
const result = datasetError ? {
|
|
147
|
+
id: rowId,
|
|
148
|
+
assertions: [],
|
|
149
|
+
passed: false,
|
|
150
|
+
error: datasetError
|
|
151
|
+
} : await runOne(d, rowId, def, rows[r], runId, options);
|
|
152
|
+
results.push(result);
|
|
153
|
+
emit({
|
|
154
|
+
type: "result",
|
|
155
|
+
result,
|
|
156
|
+
index,
|
|
157
|
+
total
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
}
|
|
70
161
|
/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */
|
|
71
162
|
async function maybeConfigureJudge(options) {
|
|
72
163
|
if (!options.judge) return;
|
|
@@ -147,23 +238,11 @@ async function runEvalsInDir(options) {
|
|
|
147
238
|
runId
|
|
148
239
|
});
|
|
149
240
|
}
|
|
150
|
-
const results = await mapPool(discovered, options.concurrency ?? DEFAULT_CONCURRENCY, async (d, index) => {
|
|
151
|
-
const
|
|
152
|
-
emit
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
index,
|
|
156
|
-
total
|
|
157
|
-
});
|
|
158
|
-
const result = await runOne(d, id, runId, options);
|
|
159
|
-
emit({
|
|
160
|
-
type: "result",
|
|
161
|
-
result,
|
|
162
|
-
index,
|
|
163
|
-
total
|
|
164
|
-
});
|
|
165
|
-
return result;
|
|
166
|
-
});
|
|
241
|
+
const results = (await mapPool(discovered, options.concurrency ?? DEFAULT_CONCURRENCY, async (d, index) => {
|
|
242
|
+
const fileResults = [];
|
|
243
|
+
await runDiscovered(d, index, total, runId, options, emit, fileResults);
|
|
244
|
+
return fileResults;
|
|
245
|
+
})).flat();
|
|
167
246
|
const summary = { results };
|
|
168
247
|
const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);
|
|
169
248
|
if (mlflow) summary.mlflow = mlflow;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"run-evals.js","names":[],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport { type DiscoveredEval, discoverEvalFiles } from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */\n timeoutMs?: number;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n * Uses tsx's programmatic loader so TypeScript eval files run without a build\n * step. The specifier is indirected so the type checker doesn't try to resolve\n * tsx's internal entry.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n\n const mod = await tsImport(pathToFileURL(file).href, import.meta.url);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Load and run a single discovered eval. Never throws — a load/run failure\n * becomes a non-passing {@link EvalResult} so one bad eval can't abort the run.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n const def = await loadEval(d.file);\n const driver = createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n timeoutMs: options.timeoutMs,\n });\n return await runEval(def, { id, driver, strict: options.strict });\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n const total = discovered.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run evals through a bounded pool so independent turns overlap instead of\n // summing their latencies. runOne never throws, so a pool worker never\n // rejects; results preserve discovery order (mapPool writes by index).\n const results = await mapPool(\n discovered,\n options.concurrency ?? DEFAULT_CONCURRENCY,\n async (d, index) => {\n const id = `${d.agent}/${d.id}`;\n emit({ type: \"start\", id, index, total });\n const result = await runOne(d, id, runId, options);\n emit({ type: \"result\", result, index, total });\n return result;\n },\n );\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;AAyEA,eAAe,SAAS,MAAuC;CAC7D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;CAIH,MAAM,MAAM,mBADA,MAAM,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI,CAClC;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;;AAST,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;AASrD,eAAe,OACb,GACA,IACA,OACA,SACqB;AACrB,KAAI;EACF,MAAM,MAAM,MAAM,SAAS,EAAE,KAAK;AAQlC,SAAO,MAAM,QAAQ,KAAK;GAAE;GAAI,QAPjB,iBAAiB;IAC9B,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IACb,WAAW,QAAQ;IACpB,CAAC;GACsC,QAAQ,QAAQ;GAAQ,CAAC;UAC1D,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;AAKL,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;AAO5B,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CACvC,MAAM,QAAQ,WAAW;AACzB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAMtC,MAAM,UAAU,MAAM,QACpB,YACA,QAAQ,eAAe,qBACvB,OAAO,GAAG,UAAU;GAClB,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;AAC3B,QAAK;IAAE,MAAM;IAAS;IAAI;IAAO;IAAO,CAAC;GACzC,MAAM,SAAS,MAAM,OAAO,GAAG,IAAI,OAAO,QAAQ;AAClD,QAAK;IAAE,MAAM;IAAU;IAAQ;IAAO;IAAO,CAAC;AAC9C,UAAO;IAEV;EAED,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
|
|
1
|
+
{"version":3,"file":"run-evals.js","names":[],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport { type DiscoveredEval, discoverEvalFiles } from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */\n timeoutMs?: number;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n * Uses tsx's programmatic loader so TypeScript eval files run without a build\n * step. The specifier is indirected so the type checker doesn't try to resolve\n * tsx's internal entry.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n\n const mod = await tsImport(pathToFileURL(file).href, import.meta.url);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // A fresh driver per row: each row is an independent conversation whose\n // thread must not carry over the previous row's history. (Multiple\n // `t.send`s within one row still share the thread — the driver's behavior.)\n const driver = createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n timeoutMs: options.timeoutMs,\n });\n return await runEval(def, { id, driver, strict: options.strict, row });\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Load one discovered eval and run it, expanding a dataset-driven eval into one\n * run per row. Appends one result per row to `results`, emitting `start`/\n * `result` around each. Never throws: a load or dataset-read failure surfaces as\n * a non-passing result. `total` counts eval files, not rows — per-row detail is\n * carried in the result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n const total = discovered.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each eval through the bounded pool — one in-flight stream per eval, so\n // the pool respects the server's per-user stream cap (see mapPool/concurrency).\n // A dataset eval expands into per-row runs that execute serially within its\n // slot; results preserve discovery order (mapPool writes by index) and row\n // order within each file. `total` counts eval files, not dataset rows — per-row\n // detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n discovered,\n options.concurrency ?? DEFAULT_CONCURRENCY,\n async (d, index) => {\n const fileResults: EvalResult[] = [];\n await runDiscovered(d, index, total, runId, options, emit, fileResults);\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAkFA,eAAe,SAAS,MAAuC;CAC7D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;CAIH,MAAM,MAAM,mBADA,MAAM,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI,CAClC;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;;AAST,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAWF,SAAO,MAAM,QAAQ,KAAK;GAAE;GAAI,QAPjB,iBAAiB;IAC9B,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IACb,WAAW,QAAQ;IACpB,CAAC;GACsC,QAAQ,QAAQ;GAAQ;GAAK,CAAC;UAC/D,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;AAWL,eAAe,cACb,GACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;CAE3B,IAAI;AACJ,KAAI;AACF,QAAM,MAAM,SAAS,EAAE,KAAK;UACrB,KAAK;AACZ,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;AAO5B,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CACvC,MAAM,QAAQ,WAAW;AACzB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkBtC,MAAM,WATU,MAAM,QACpB,YACA,QAAQ,eAAe,qBACvB,OAAO,GAAG,UAAU;GAClB,MAAM,cAA4B,EAAE;AACpC,SAAM,cAAc,GAAG,OAAO,OAAO,OAAO,SAAS,MAAM,YAAY;AACvE,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
|
package/dist/evals/types.d.ts
CHANGED
|
@@ -38,7 +38,11 @@ interface AssertionHandle {
|
|
|
38
38
|
gate(): AssertionHandle;
|
|
39
39
|
/** Demote to a tracked metric — doesn't fail unless running with `strict`. */
|
|
40
40
|
soft(): AssertionHandle;
|
|
41
|
-
/**
|
|
41
|
+
/**
|
|
42
|
+
* Set the pass threshold for a scored assertion: it passes only when the
|
|
43
|
+
* score is at least `threshold`. Keeps the current severity (gate unless also
|
|
44
|
+
* chained with `.soft()`).
|
|
45
|
+
*/
|
|
42
46
|
atLeast(threshold: number): AssertionHandle;
|
|
43
47
|
}
|
|
44
48
|
/** What a driver returns for a single `t.send`. */
|
|
@@ -60,17 +64,38 @@ interface DriveResult {
|
|
|
60
64
|
*/
|
|
61
65
|
interface EvalDriver {
|
|
62
66
|
send(message: string): Promise<DriveResult>;
|
|
67
|
+
/**
|
|
68
|
+
* Drop the current conversation so the next `send` starts a fresh thread.
|
|
69
|
+
* Optional: drivers without a session concept omit it.
|
|
70
|
+
*/
|
|
71
|
+
reset?(): void;
|
|
63
72
|
}
|
|
64
73
|
/** The `t` context passed to an eval's `test` function. */
|
|
65
74
|
interface TestContext {
|
|
66
75
|
/** Send a user message to the agent and capture its response. */
|
|
67
76
|
send(message: string): Promise<void>;
|
|
77
|
+
/**
|
|
78
|
+
* Start a fresh conversation: the next `send` opens a new thread with no
|
|
79
|
+
* history. Use to run several independent one-shot checks in one test.
|
|
80
|
+
* Consecutive `send`s (without a `reset`) stay in one multi-turn conversation.
|
|
81
|
+
*/
|
|
82
|
+
reset(): void;
|
|
68
83
|
/** The last assistant reply. */
|
|
69
84
|
readonly reply: string;
|
|
70
85
|
/** Tools called during the last turn. */
|
|
71
86
|
readonly toolCalls: string[];
|
|
72
87
|
/** The current session/thread id, if any. */
|
|
73
88
|
readonly sessionId: string | undefined;
|
|
89
|
+
/**
|
|
90
|
+
* The current dataset row's `inputs` when the eval is dataset-driven (see
|
|
91
|
+
* {@link EvalDefinition.dataset}); `{}` for a plain single-run eval.
|
|
92
|
+
*/
|
|
93
|
+
readonly input: Record<string, unknown>;
|
|
94
|
+
/**
|
|
95
|
+
* The current dataset row's `expectations` (ground truth / guidelines), or
|
|
96
|
+
* `undefined` when the row has none or the eval isn't dataset-driven.
|
|
97
|
+
*/
|
|
98
|
+
readonly expected: Record<string, unknown> | undefined;
|
|
74
99
|
/** Assert the last turn completed successfully (gate by default). */
|
|
75
100
|
succeeded(): AssertionHandle;
|
|
76
101
|
/** Assert a tool was called during the run (gate by default). */
|
|
@@ -79,9 +104,10 @@ interface TestContext {
|
|
|
79
104
|
check(value: string, matcher: Matcher): AssertionHandle;
|
|
80
105
|
/**
|
|
81
106
|
* LLM-as-judge scoring of the last reply (via autoevals → a Databricks judge
|
|
82
|
-
* model). Each returns a scored
|
|
83
|
-
* to
|
|
84
|
-
* judge to be configured
|
|
107
|
+
* model). Each returns a scored assertion that gates by default (a miss fails
|
|
108
|
+
* the eval); chain `.atLeast(n)` to change the pass threshold or `.soft()` to
|
|
109
|
+
* demote to a tracked-only metric. Requires the judge to be configured
|
|
110
|
+
* (`--judge-model`).
|
|
85
111
|
*/
|
|
86
112
|
judge: {
|
|
87
113
|
/** Score factuality of the reply against an expected reference. */factuality(expected: string): Promise<AssertionHandle>; /** Score whether the reply answers the question, per optional `criteria`. */
|
|
@@ -103,6 +129,16 @@ interface EvalDefinition {
|
|
|
103
129
|
description?: string;
|
|
104
130
|
/** Target agent id. Defaults to the eval's parent `server/agents/<id>` dir. */
|
|
105
131
|
agent?: string;
|
|
132
|
+
/**
|
|
133
|
+
* Run this eval once per row of a Databricks managed evaluation dataset (a
|
|
134
|
+
* Unity Catalog `catalog.schema.table` with `inputs`/`expectations` columns).
|
|
135
|
+
* Each row is bound to `t.input`/`t.expected`. Requires the runner to have a
|
|
136
|
+
* workspace client + warehouse (`--warehouse-id`). Omit for a single-run eval.
|
|
137
|
+
*/
|
|
138
|
+
dataset?: {
|
|
139
|
+
table: string;
|
|
140
|
+
limit?: number;
|
|
141
|
+
};
|
|
106
142
|
/** The eval body: drive the agent and assert on its behavior. */
|
|
107
143
|
test(t: TestContext): Promise<void> | void;
|
|
108
144
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;
|
|
1
|
+
{"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;EAQoB;EAN5B,IAAA,IAAQ,eAAA;EAMmC;;;;;EAA3C,OAAA,CAAQ,SAAA,WAAoB,eAAA;AAAA;;UAIb,WAAA;EAJ4B;EAM3C,KAAA;EAF0B;EAI1B,SAAA;EAJ0B;EAM1B,SAAA;EAFA;EAIA,SAAA;EAAA;EAEA,OAAA;AAAA;;AAOF;;;UAAiB,UAAA;EACf,IAAA,CAAK,OAAA,WAAkB,OAAA,CAAQ,WAAA;EAA1B;;;;EAKL,KAAA;AAAA;AAIF;AAAA,UAAiB,WAAA;;EAEf,IAAA,CAAK,OAAA,WAAkB,OAAA;EAiBP;;;;;EAXhB,KAAA;EAgCwC;EAAA,SA9B/B,KAAA;EAgC6B;EAAA,SA9B7B,SAAA;EAgCM;EAAA,SA9BN,SAAA;EA8BwB;;;;EAAA,SAzBxB,KAAA,EAAO,MAAA;EAjBO;;;;EAAA,SAsBd,QAAA,EAAU,MAAA;EALV;EAOT,SAAA,IAAa,eAAA;EAFJ;EAIT,UAAA,CAAW,IAAA,WAAe,eAAA;EAF1B;EAIA,KAAA,CAAM,KAAA,UAAe,OAAA,EAAS,OAAA,GAAU,eAAA;EAFxC;;;;;;;EAUA,KAAA;IAAA,mEAEE,UAAA,CAAW,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAA3B;IAEX,QAAA,CAAS,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAFE;IAItC,MAAA,CAAO,IAAA,EAAM,eAAA,GAAkB,OAAA,CAAQ,eAAA;EAAA;EAFX;EAK9B,IAAA,CAAK,MAAA;AAAA;;UAIU,eAAA;EACf,IAAA;EACA,cAAA;EACA,YAAA,EAAc,MAAA;AAAA;;UAIC,cAAA;EAPA;EASf,WAAA;;EAEA,KAAA;EAVA;;;;;;EAiBA,OAAA;IAAY,KAAA;IAAe,KAAA;EAAA;EAT3B;EAWA,IAAA,CAAK,CAAA,EAAG,WAAA,GAAc,OAAA;AAAA;;UAIP,UAAA;EACf,EAAA;EACA,WAAA;EANK;EAQL,OAAA;IAAY,MAAA;EAAA;EACZ,UAAA,EAAY,eAAA;EALa;EAOzB,MAAA;EAF2B;EAI3B,KAAA;EAPA;EASA,OAAA;AAAA"}
|
package/dist/index.d.ts
CHANGED
|
@@ -24,6 +24,7 @@ import { AppKitError } from "./errors/base.js";
|
|
|
24
24
|
import { AuthenticationError } from "./errors/authentication.js";
|
|
25
25
|
import { ConfigurationError } from "./errors/configuration.js";
|
|
26
26
|
import { ConnectionError } from "./errors/connection.js";
|
|
27
|
+
import { DatabaseValidationError, DatabaseValidationIssue } from "./errors/database-validation.js";
|
|
27
28
|
import { ExecutionError } from "./errors/execution.js";
|
|
28
29
|
import { InitializationError } from "./errors/initialization.js";
|
|
29
30
|
import { ServerError } from "./errors/server.js";
|
|
@@ -53,4 +54,4 @@ import "./plugins/ga-exports.generated.js";
|
|
|
53
54
|
import { extractServingEndpoints, findServerFile } from "./type-generator/serving/server-file-extractor.js";
|
|
54
55
|
import { appKitServingTypesPlugin } from "./type-generator/serving/vite-plugin.js";
|
|
55
56
|
import { appKitTypesPlugin } from "./type-generator/vite-plugin.js";
|
|
56
|
-
export { ApiError, AppKitError, AuthenticationError, type BasePluginConfig, type CacheConfig, CacheManager, type ConfigSchema, ConfigurationError, ConnectionError, type Counter, type DatabaseCredential, type DatabaseRegistry, type EndpointConfig, ExecutionError, type ExecutionResult, type FileAction, type FilePolicy, type FilePolicyUser, type FileResource, type GenerateDatabaseCredentialRequest, type Histogram, type IAppRouter, type IJobsConfig, type ITelemetry, InitializationError, type JobAPI, type JobConfig, type JobsConnectorConfig, type JobsExport, type LakebasePool, type LakebasePoolConfig, type LakebasePoolManager, Plugin, type PluginData, type PluginManifest, PolicyDeniedError, READ_ACTIONS, type RequestedClaims, RequestedClaimsPermissionSet, type RequestedResource, type ResourceEntry, type ResourceFieldEntry, type ResourcePermission, ResourceRegistry, type ResourceRequirement, ResourceType, ServerError, type ServingEndpointEntry, type ServingEndpointRegistry, type ServingFactory, SeverityNumber, type Span, SpanStatusCode, type StreamExecutionSettings, type TelemetryConfig, type ToPlugin, TunnelError, ValidationError, type ValidationResult, WRITE_ACTIONS, type WorkspaceClient, type WorkspaceClientOptions, analytics, appKitServingTypesPlugin, appKitTypesPlugin, createApp, createLakebasePool, createLakebasePoolManager, createWorkspaceClient, defineManifest, extractServingEndpoints, files, findServerFile, generateDatabaseCredential, genie, getExecutionContext, getLakebaseOrmConfig, getLakebasePgConfig, getPluginManifest, getResourceRequirements, getUsernameWithApiLookup, getWorkspaceClient, isSQLTypeMarker, jobs, lakebase, server, serving, sql, toPlugin };
|
|
57
|
+
export { ApiError, AppKitError, AuthenticationError, type BasePluginConfig, type CacheConfig, CacheManager, type ConfigSchema, ConfigurationError, ConnectionError, type Counter, type DatabaseCredential, type DatabaseRegistry, DatabaseValidationError, type DatabaseValidationIssue, type EndpointConfig, ExecutionError, type ExecutionResult, type FileAction, type FilePolicy, type FilePolicyUser, type FileResource, type GenerateDatabaseCredentialRequest, type Histogram, type IAppRouter, type IJobsConfig, type ITelemetry, InitializationError, type JobAPI, type JobConfig, type JobsConnectorConfig, type JobsExport, type LakebasePool, type LakebasePoolConfig, type LakebasePoolManager, Plugin, type PluginData, type PluginManifest, PolicyDeniedError, READ_ACTIONS, type RequestedClaims, RequestedClaimsPermissionSet, type RequestedResource, type ResourceEntry, type ResourceFieldEntry, type ResourcePermission, ResourceRegistry, type ResourceRequirement, ResourceType, ServerError, type ServingEndpointEntry, type ServingEndpointRegistry, type ServingFactory, SeverityNumber, type Span, SpanStatusCode, type StreamExecutionSettings, type TelemetryConfig, type ToPlugin, TunnelError, ValidationError, type ValidationResult, WRITE_ACTIONS, type WorkspaceClient, type WorkspaceClientOptions, analytics, appKitServingTypesPlugin, appKitTypesPlugin, createApp, createLakebasePool, createLakebasePoolManager, createWorkspaceClient, defineManifest, extractServingEndpoints, files, findServerFile, generateDatabaseCredential, genie, getExecutionContext, getLakebaseOrmConfig, getLakebasePgConfig, getPluginManifest, getResourceRequirements, getUsernameWithApiLookup, getWorkspaceClient, isSQLTypeMarker, jobs, lakebase, server, serving, sql, toPlugin };
|
package/dist/index.js
CHANGED
|
@@ -6,6 +6,7 @@ import { AppKitError } from "./errors/base.js";
|
|
|
6
6
|
import { AuthenticationError } from "./errors/authentication.js";
|
|
7
7
|
import { ConfigurationError } from "./errors/configuration.js";
|
|
8
8
|
import { ConnectionError } from "./errors/connection.js";
|
|
9
|
+
import { DatabaseValidationError } from "./errors/database-validation.js";
|
|
9
10
|
import { ExecutionError } from "./errors/execution.js";
|
|
10
11
|
import { InitializationError } from "./errors/initialization.js";
|
|
11
12
|
import { ServerError } from "./errors/server.js";
|
|
@@ -39,4 +40,4 @@ import { server } from "./plugins/server/index.js";
|
|
|
39
40
|
import { serving } from "./plugins/serving/serving.js";
|
|
40
41
|
import "./plugins/ga-exports.generated.js";
|
|
41
42
|
|
|
42
|
-
export { ApiError, AppKitError, AuthenticationError, CacheManager, ConfigurationError, ConnectionError, ExecutionError, InitializationError, Plugin, PolicyDeniedError, READ_ACTIONS, RequestedClaimsPermissionSet, ResourceRegistry, ResourceType, ServerError, SeverityNumber, SpanStatusCode, TunnelError, ValidationError, WRITE_ACTIONS, analytics, appKitServingTypesPlugin, appKitTypesPlugin, createApp, createLakebasePool, createLakebasePoolManager, createWorkspaceClient, defineManifest, extractServingEndpoints, files, findServerFile, generateDatabaseCredential, genie, getExecutionContext, getLakebaseOrmConfig, getLakebasePgConfig, getPluginManifest, getResourceRequirements, getUsernameWithApiLookup, getWorkspaceClient, isSQLTypeMarker, jobs, lakebase, server, serving, sql, toPlugin };
|
|
43
|
+
export { ApiError, AppKitError, AuthenticationError, CacheManager, ConfigurationError, ConnectionError, DatabaseValidationError, ExecutionError, InitializationError, Plugin, PolicyDeniedError, READ_ACTIONS, RequestedClaimsPermissionSet, ResourceRegistry, ResourceType, ServerError, SeverityNumber, SpanStatusCode, TunnelError, ValidationError, WRITE_ACTIONS, analytics, appKitServingTypesPlugin, appKitTypesPlugin, createApp, createLakebasePool, createLakebasePoolManager, createWorkspaceClient, defineManifest, extractServingEndpoints, files, findServerFile, generateDatabaseCredential, genie, getExecutionContext, getLakebaseOrmConfig, getLakebasePgConfig, getPluginManifest, getResourceRequirements, getUsernameWithApiLookup, getWorkspaceClient, isSQLTypeMarker, jobs, lakebase, server, serving, sql, toPlugin };
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"plugin.d.ts","names":[],"sources":["../../src/plugin/plugin.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;uBA0NsB,MAAA,iBACJ,gBAAA,GAAmB,gBAAA,aACxB,UAAA;EAAA,UA4BW,MAAA,EAAQ,OAAA;EAAA,UA3BpB,OAAA;EAAA,UACA,KAAA,EAAQ,YAAA;EAAA,UACR,GAAA,EAAK,UAAA;EAAA,UACL,aAAA,EAAe,aAAA;EAAA,UACf,aAAA,EAAe,aAAA;EAAA,UACf,SAAA,EAAY,UAAA;EAAA,UACZ,OAAA,GAAU,aAAA;
|
|
1
|
+
{"version":3,"file":"plugin.d.ts","names":[],"sources":["../../src/plugin/plugin.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;uBA0NsB,MAAA,iBACJ,gBAAA,GAAmB,gBAAA,aACxB,UAAA;EAAA,UA4BW,MAAA,EAAQ,OAAA;EAAA,UA3BpB,OAAA;EAAA,UACA,KAAA,EAAQ,YAAA;EAAA,UACR,GAAA,EAAK,UAAA;EAAA,UACL,aAAA,EAAe,aAAA;EAAA,UACf,aAAA,EAAe,aAAA;EAAA,UACf,SAAA,EAAY,UAAA;EAAA,UACZ,OAAA,GAAU,aAAA;EA0SO;EAAA,QAvSnB,mBAAA;EAwSG;EAAA,QArSH,oBAAA;EAsSN;;;;;;EAAA,OA9RK,KAAA,EAAO,WAAA;EA+W0B;;;EA1WxC,IAAA;cAEsB,MAAA,EAAQ,OAAA;EAAA,QAoBtB,gBAAA;EAuVG;;;;;;;;EAlUX,aAAA,CACE,IAAA;IACE,OAAA;IACA,eAAA,GAAkB,gBAAA;EAAA;EAgBtB,YAAA,CAAa,CAAA,EAAG,OAAA,CAAQ,MAAA;EAIlB,KAAA,CAAA,GAAK,OAAA;EAEX,YAAA,CAAA,GAAgB,iBAAA;EAIhB,uBAAA,CAAA,GAA2B,WAAA;EAI3B,qBAAA,CAAA;EAgbyB;;;;;;;;;;;;;;;;;;;;;;;;EApZzB,OAAA,CAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAiDA,YAAA,CAAA,GAAgB,MAAA;;;;;;;;;;;YAcN,aAAA,CAAc,GAAA,EAAK,OAAA,CAAQ,OAAA;;;;;;;;;;;EAiBrC,MAAA,CAAO,GAAA,EAAK,OAAA,CAAQ,OAAA;;;;;;;;;;;;;;UAyDZ,kBAAA;EAAA,UAoCQ,aAAA,GAAA,CACd,GAAA,EAAK,YAAA,EACL,EAAA,EAAI,oBAAA,CAAqB,CAAA,GACzB,OAAA,EAAS,uBAAA,EACT,OAAA,YAAgB,OAAA;;;;;;;;;;YAgFF,OAAA,GAAA,CACd,EAAA,GAAK,MAAA,GAAS,WAAA,KAAgB,OAAA,CAAQ,CAAA,GACtC,OAAA,EAAS,uBAAA,EACT,OAAA,YACC,OAAA,CAAQ,eAAA,CAAgB,CAAA;EAAA,UAmDjB,gBAAA,CAAiB,IAAA,UAAc,IAAA;EAAA,UAI/B,KAAA,YAAA,CACR,MAAA,EAAQ,OAAA,CAAQ,MAAA,EAChB,MAAA,EAAQ,WAAA;EAAA,QAeF,qBAAA;EAAA,QAaA,kBAAA;EAAA,QAqCM,wBAAA;EAAA,QAqBN,iBAAA;AAAA"}
|
package/dist/plugin/plugin.js
CHANGED
|
@@ -382,7 +382,7 @@ var Plugin = class {
|
|
|
382
382
|
const raw = value.call(target);
|
|
383
383
|
if (raw == null) return {};
|
|
384
384
|
if (typeof raw === "function") return raw;
|
|
385
|
-
if (isPlainObject(raw)) return wrapExportFunctions(raw, wrapCall);
|
|
385
|
+
if (isPlainObject(raw)) return wrapExportFunctions(raw, (fn) => wrapCall(fn.bind(target)));
|
|
386
386
|
return raw;
|
|
387
387
|
};
|
|
388
388
|
return wrapCall(value.bind(target));
|