@databricks/appkit 0.71.0 → 0.73.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (191) hide show
  1. package/CLAUDE.md +63 -0
  2. package/NOTICE.md +3 -2
  3. package/dist/appkit/package.js +1 -1
  4. package/dist/beta.d.ts +18 -3
  5. package/dist/beta.js +14 -1
  6. package/dist/cli/commands/agent/eval.js +120 -0
  7. package/dist/cli/commands/agent/eval.js.map +1 -0
  8. package/dist/cli/commands/agent/index.js +18 -0
  9. package/dist/cli/commands/agent/index.js.map +1 -0
  10. package/dist/cli/index.js +2 -0
  11. package/dist/cli/index.js.map +1 -1
  12. package/dist/connectors/index.js +2 -0
  13. package/dist/connectors/mlflow/auth.d.ts +28 -0
  14. package/dist/connectors/mlflow/auth.d.ts.map +1 -0
  15. package/dist/connectors/mlflow/auth.js +70 -0
  16. package/dist/connectors/mlflow/auth.js.map +1 -0
  17. package/dist/connectors/mlflow/client.d.ts +51 -0
  18. package/dist/connectors/mlflow/client.d.ts.map +1 -0
  19. package/dist/connectors/mlflow/client.js +93 -0
  20. package/dist/connectors/mlflow/client.js.map +1 -0
  21. package/dist/connectors/mlflow/index.d.ts +2 -0
  22. package/dist/database/errors.js +15 -5
  23. package/dist/database/errors.js.map +1 -1
  24. package/dist/database/runtime/data-path.d.ts +7 -0
  25. package/dist/database/runtime/data-path.d.ts.map +1 -0
  26. package/dist/database/runtime/data-path.js.map +1 -1
  27. package/dist/database/runtime/engine/drizzle-data-path.js +7 -5
  28. package/dist/database/runtime/engine/drizzle-data-path.js.map +1 -1
  29. package/dist/database/schema-builder/define-schema.d.ts +1 -1
  30. package/dist/database/schema-builder/define-schema.js +1 -1
  31. package/dist/database/schema-builder/define-schema.js.map +1 -1
  32. package/dist/errors/database-validation.d.ts +23 -0
  33. package/dist/errors/database-validation.d.ts.map +1 -0
  34. package/dist/errors/database-validation.js +24 -0
  35. package/dist/errors/database-validation.js.map +1 -0
  36. package/dist/errors/index.js +1 -0
  37. package/dist/evals/dataset.d.ts +36 -0
  38. package/dist/evals/dataset.d.ts.map +1 -0
  39. package/dist/evals/dataset.js +36 -0
  40. package/dist/evals/dataset.js.map +1 -0
  41. package/dist/evals/define-eval.d.ts +26 -0
  42. package/dist/evals/define-eval.d.ts.map +1 -0
  43. package/dist/evals/define-eval.js +28 -0
  44. package/dist/evals/define-eval.js.map +1 -0
  45. package/dist/evals/discover.d.ts +20 -0
  46. package/dist/evals/discover.d.ts.map +1 -0
  47. package/dist/evals/discover.js +49 -0
  48. package/dist/evals/discover.js.map +1 -0
  49. package/dist/evals/http-driver.d.ts +33 -0
  50. package/dist/evals/http-driver.d.ts.map +1 -0
  51. package/dist/evals/http-driver.js +123 -0
  52. package/dist/evals/http-driver.js.map +1 -0
  53. package/dist/evals/index.d.ts +14 -0
  54. package/dist/evals/index.js +14 -0
  55. package/dist/evals/judge.d.ts +27 -0
  56. package/dist/evals/judge.d.ts.map +1 -0
  57. package/dist/evals/judge.js +77 -0
  58. package/dist/evals/judge.js.map +1 -0
  59. package/dist/evals/matchers.d.ts +12 -0
  60. package/dist/evals/matchers.d.ts.map +1 -0
  61. package/dist/evals/matchers.js +26 -0
  62. package/dist/evals/matchers.js.map +1 -0
  63. package/dist/evals/mlflow-report.d.ts +37 -0
  64. package/dist/evals/mlflow-report.d.ts.map +1 -0
  65. package/dist/evals/mlflow-report.js +161 -0
  66. package/dist/evals/mlflow-report.js.map +1 -0
  67. package/dist/evals/mlflow-run.d.ts +13 -0
  68. package/dist/evals/mlflow-run.d.ts.map +1 -0
  69. package/dist/evals/mlflow-run.js +101 -0
  70. package/dist/evals/mlflow-run.js.map +1 -0
  71. package/dist/evals/pool.js +24 -0
  72. package/dist/evals/pool.js.map +1 -0
  73. package/dist/evals/report.d.ts +25 -0
  74. package/dist/evals/report.d.ts.map +1 -0
  75. package/dist/evals/report.js +57 -0
  76. package/dist/evals/report.js.map +1 -0
  77. package/dist/evals/run-eval.d.ts +23 -0
  78. package/dist/evals/run-eval.d.ts.map +1 -0
  79. package/dist/evals/run-eval.js +152 -0
  80. package/dist/evals/run-eval.js.map +1 -0
  81. package/dist/evals/run-evals.d.ts +94 -0
  82. package/dist/evals/run-evals.d.ts.map +1 -0
  83. package/dist/evals/run-evals.js +257 -0
  84. package/dist/evals/run-evals.js.map +1 -0
  85. package/dist/evals/types.d.ts +163 -0
  86. package/dist/evals/types.d.ts.map +1 -0
  87. package/dist/index.d.ts +2 -1
  88. package/dist/index.js +2 -1
  89. package/dist/plugin/plugin.d.ts.map +1 -1
  90. package/dist/plugin/plugin.js +1 -1
  91. package/dist/plugin/plugin.js.map +1 -1
  92. package/dist/plugins/agents/agents.js +1 -1
  93. package/dist/plugins/database/crud/contract.js +17 -8
  94. package/dist/plugins/database/crud/contract.js.map +1 -1
  95. package/dist/plugins/database/crud/exposure.js +63 -22
  96. package/dist/plugins/database/crud/exposure.js.map +1 -1
  97. package/dist/plugins/database/crud/request.js +50 -0
  98. package/dist/plugins/database/crud/request.js.map +1 -0
  99. package/dist/plugins/database/crud/response.js +77 -0
  100. package/dist/plugins/database/crud/response.js.map +1 -0
  101. package/dist/plugins/database/crud/routes.js +71 -52
  102. package/dist/plugins/database/crud/routes.js.map +1 -1
  103. package/dist/plugins/database/database.d.ts +6 -4
  104. package/dist/plugins/database/database.d.ts.map +1 -1
  105. package/dist/plugins/database/database.js +46 -16
  106. package/dist/plugins/database/database.js.map +1 -1
  107. package/dist/plugins/database/defaults.js +5 -1
  108. package/dist/plugins/database/defaults.js.map +1 -1
  109. package/dist/plugins/database/entity-client.js +143 -10
  110. package/dist/plugins/database/entity-client.js.map +1 -1
  111. package/dist/plugins/database/entity-types.d.ts +1 -1
  112. package/dist/plugins/database/hooks.d.ts +38 -0
  113. package/dist/plugins/database/hooks.d.ts.map +1 -0
  114. package/dist/plugins/database/index.d.ts +3 -2
  115. package/dist/plugins/database/lifecycle.js +67 -28
  116. package/dist/plugins/database/lifecycle.js.map +1 -1
  117. package/dist/plugins/database/scope.js +58 -0
  118. package/dist/plugins/database/scope.js.map +1 -0
  119. package/dist/plugins/database/types.d.ts +40 -12
  120. package/dist/plugins/database/types.d.ts.map +1 -1
  121. package/docs/api/appkit/Class.AppKitError.md +1 -0
  122. package/docs/api/appkit/Class.DatabaseValidationError.md +191 -0
  123. package/docs/api/appkit/Class.MlflowClient.md +103 -0
  124. package/docs/api/appkit/Function.buildAssessments.md +16 -0
  125. package/docs/api/appkit/Function.configureJudge.md +18 -0
  126. package/docs/api/appkit/Function.createHttpDriver.md +18 -0
  127. package/docs/api/appkit/Function.defineEval.md +35 -0
  128. package/docs/api/appkit/Function.defineSchema.md +1 -1
  129. package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
  130. package/docs/api/appkit/Function.equals.md +18 -0
  131. package/docs/api/appkit/Function.evalGlyph.md +18 -0
  132. package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
  133. package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
  134. package/docs/api/appkit/Function.formatEvalResults.md +18 -0
  135. package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
  136. package/docs/api/appkit/Function.includes.md +18 -0
  137. package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
  138. package/docs/api/appkit/Function.matches.md +18 -0
  139. package/docs/api/appkit/Function.normalizeHost.md +18 -0
  140. package/docs/api/appkit/Function.readEvalDataset.md +21 -0
  141. package/docs/api/appkit/Function.reportToMlflow.md +23 -0
  142. package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
  143. package/docs/api/appkit/Function.resolveWorkspaceClient.md +18 -0
  144. package/docs/api/appkit/Function.runEval.md +19 -0
  145. package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
  146. package/docs/api/appkit/Function.summarize.md +16 -0
  147. package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
  148. package/docs/api/appkit/Interface.AssertionResult.md +48 -0
  149. package/docs/api/appkit/Interface.Assessment.md +83 -0
  150. package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
  151. package/docs/api/appkit/Interface.DatabaseValidationIssue.md +21 -0
  152. package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
  153. package/docs/api/appkit/Interface.DatasetRow.md +21 -0
  154. package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
  155. package/docs/api/appkit/Interface.DriveResult.md +58 -0
  156. package/docs/api/appkit/Interface.EntityMutationHooks.md +173 -0
  157. package/docs/api/appkit/Interface.EvalDefinition.md +74 -0
  158. package/docs/api/appkit/Interface.EvalDriver.md +37 -0
  159. package/docs/api/appkit/Interface.EvalResult.md +83 -0
  160. package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
  161. package/docs/api/appkit/Interface.EvalSummary.md +48 -0
  162. package/docs/api/appkit/Interface.HookApp.md +12 -0
  163. package/docs/api/appkit/Interface.HookContext.md +21 -0
  164. package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
  165. package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
  166. package/docs/api/appkit/Interface.JudgeScore.md +21 -0
  167. package/docs/api/appkit/Interface.MatchResult.md +34 -0
  168. package/docs/api/appkit/Interface.PostResult.md +30 -0
  169. package/docs/api/appkit/Interface.ReadEvalDatasetOptions.md +34 -0
  170. package/docs/api/appkit/Interface.ReadSerializerContext.md +21 -0
  171. package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
  172. package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
  173. package/docs/api/appkit/Interface.RunEvalOptions.md +45 -0
  174. package/docs/api/appkit/Interface.RunEvalsOptions.md +214 -0
  175. package/docs/api/appkit/Interface.TestContext.md +245 -0
  176. package/docs/api/appkit/TypeAlias.DatabaseApiConfig.md +53 -0
  177. package/docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md +8 -0
  178. package/docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md +49 -0
  179. package/docs/api/appkit/TypeAlias.DatabaseExports.md +3 -3
  180. package/docs/api/appkit/TypeAlias.EntityHooks.md +25 -0
  181. package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
  182. package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +16 -5
  183. package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
  184. package/docs/api/appkit/TypeAlias.ReadSerializer.md +19 -0
  185. package/docs/api/appkit/TypeAlias.Severity.md +8 -0
  186. package/docs/api/appkit/TypeAlias.TransactionClient.md +19 -0
  187. package/docs/api/appkit.md +157 -95
  188. package/docs/plugins/database.md +144 -0
  189. package/llms.txt +63 -0
  190. package/package.json +3 -2
  191. package/sbom.cdx.json +1 -1
@@ -0,0 +1 @@
1
+ {"version":3,"file":"report.js","names":[],"sources":["../../src/evals/report.ts"],"sourcesContent":["import type { EvalResult } from \"./types\";\n\nexport interface EvalSummary {\n total: number;\n passed: number;\n failed: number;\n skipped: number;\n /** True when no eval failed (skips don't count as failures). */\n allPassed: boolean;\n}\n\nexport function summarize(results: EvalResult[]): EvalSummary {\n let passed = 0;\n let failed = 0;\n let skipped = 0;\n for (const r of results) {\n if (r.skipped) skipped++;\n else if (r.passed) passed++;\n else failed++;\n }\n return {\n total: results.length,\n passed,\n failed,\n skipped,\n allPassed: failed === 0,\n };\n}\n\n/** Status glyph for a single eval result. */\nexport function evalGlyph(result: EvalResult): string {\n if (result.skipped) return \"−\";\n return result.passed ? \"✓\" : \"✗\";\n}\n\n/** The one-line header for a single eval result (no failure detail). */\nexport function formatEvalHeadline(result: EvalResult): string {\n if (result.skipped) {\n return `− ${result.id} (skipped${\n result.skipped.reason ? `: ${result.skipped.reason}` : \"\"\n })`;\n }\n return `${evalGlyph(result)} ${result.id}${\n result.description ? ` — ${result.description}` : \"\"\n }`;\n}\n\n/** Indented detail lines for a failing eval (error + failing assertions). */\nexport function formatEvalDetail(result: EvalResult): string[] {\n const lines: string[] = [];\n if (result.error) lines.push(` error: ${result.error}`);\n for (const a of result.assertions) {\n if (a.pass) continue;\n const tag = a.severity === \"soft\" ? \"soft\" : \"gate\";\n lines.push(` ✗ [${tag}] ${a.label}${a.detail ? ` — ${a.detail}` : \"\"}`);\n }\n return lines;\n}\n\n/** The final PASS/FAIL summary line. */\nexport function formatSummaryLine(results: EvalResult[]): string {\n const s = summarize(results);\n return `${s.allPassed ? \"PASS\" : \"FAIL\"} — ${s.passed} passed, ${s.failed} failed, ${s.skipped} skipped (${s.total} total)`;\n}\n\n/** Render all results as a human-readable console report (non-streaming). */\nexport function formatEvalResults(results: EvalResult[]): string {\n const lines: string[] = [];\n for (const r of results) {\n lines.push(formatEvalHeadline(r));\n lines.push(...formatEvalDetail(r));\n }\n lines.push(\"\");\n lines.push(formatSummaryLine(results));\n return lines.join(\"\\n\");\n}\n"],"mappings":";AAWA,SAAgB,UAAU,SAAoC;CAC5D,IAAI,SAAS;CACb,IAAI,SAAS;CACb,IAAI,UAAU;AACd,MAAK,MAAM,KAAK,QACd,KAAI,EAAE,QAAS;UACN,EAAE,OAAQ;KACd;AAEP,QAAO;EACL,OAAO,QAAQ;EACf;EACA;EACA;EACA,WAAW,WAAW;EACvB;;;AAIH,SAAgB,UAAU,QAA4B;AACpD,KAAI,OAAO,QAAS,QAAO;AAC3B,QAAO,OAAO,SAAS,MAAM;;;AAI/B,SAAgB,mBAAmB,QAA4B;AAC7D,KAAI,OAAO,QACT,QAAO,KAAK,OAAO,GAAG,WACpB,OAAO,QAAQ,SAAS,KAAK,OAAO,QAAQ,WAAW,GACxD;AAEH,QAAO,GAAG,UAAU,OAAO,CAAC,GAAG,OAAO,KACpC,OAAO,cAAc,MAAM,OAAO,gBAAgB;;;AAKtD,SAAgB,iBAAiB,QAA8B;CAC7D,MAAM,QAAkB,EAAE;AAC1B,KAAI,OAAO,MAAO,OAAM,KAAK,cAAc,OAAO,QAAQ;AAC1D,MAAK,MAAM,KAAK,OAAO,YAAY;AACjC,MAAI,EAAE,KAAM;EACZ,MAAM,MAAM,EAAE,aAAa,SAAS,SAAS;AAC7C,QAAM,KAAK,UAAU,IAAI,IAAI,EAAE,QAAQ,EAAE,SAAS,MAAM,EAAE,WAAW,KAAK;;AAE5E,QAAO;;;AAIT,SAAgB,kBAAkB,SAA+B;CAC/D,MAAM,IAAI,UAAU,QAAQ;AAC5B,QAAO,GAAG,EAAE,YAAY,SAAS,OAAO,KAAK,EAAE,OAAO,WAAW,EAAE,OAAO,WAAW,EAAE,QAAQ,YAAY,EAAE,MAAM;;;AAIrH,SAAgB,kBAAkB,SAA+B;CAC/D,MAAM,QAAkB,EAAE;AAC1B,MAAK,MAAM,KAAK,SAAS;AACvB,QAAM,KAAK,mBAAmB,EAAE,CAAC;AACjC,QAAM,KAAK,GAAG,iBAAiB,EAAE,CAAC;;AAEpC,OAAM,KAAK,GAAG;AACd,OAAM,KAAK,kBAAkB,QAAQ,CAAC;AACtC,QAAO,MAAM,KAAK,KAAK"}
@@ -0,0 +1,23 @@
1
+ import { DatasetRow } from "./dataset.js";
2
+ import { EvalDefinition, EvalDriver, EvalResult } from "./types.js";
3
+
4
+ //#region src/evals/run-eval.d.ts
5
+ interface RunEvalOptions {
6
+ /** Stable id for the eval (e.g. its file path relative to the evals dir). */
7
+ id: string;
8
+ /** Drives the agent and returns reply/tool-calls/success per `send`. */
9
+ driver: EvalDriver;
10
+ /** When true, soft assertion failures also fail the eval. */
11
+ strict?: boolean;
12
+ /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */
13
+ row?: DatasetRow;
14
+ }
15
+ /**
16
+ * Run a single eval against a driver. Never throws for assertion or agent
17
+ * failures — those become a non-passing {@link EvalResult}. Only a malformed
18
+ * eval definition surfaces as `result.error`.
19
+ */
20
+ declare function runEval(def: EvalDefinition, options: RunEvalOptions): Promise<EvalResult>;
21
+ //#endregion
22
+ export { RunEvalOptions, runEval };
23
+ //# sourceMappingURL=run-eval.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run-eval.d.ts","names":[],"sources":["../../src/evals/run-eval.ts"],"mappings":";;;;UAuBiB,cAAA;;EAEf,EAAA;EAF6B;EAI7B,MAAA,EAAQ,UAAA;EAIQ;EAFhB,MAAA;EAFA;EAIA,GAAA,GAAM,UAAA;AAAA;;;;;AAQR;iBAAsB,OAAA,CACpB,GAAA,EAAK,cAAA,EACL,OAAA,EAAS,cAAA,GACR,OAAA,CAAQ,UAAA"}
@@ -0,0 +1,152 @@
1
+ import { judgeClosedQA, judgeCustom, judgeFactuality } from "./judge.js";
2
+
3
+ //#region src/evals/run-eval.ts
4
+ /** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */
5
+ const DEFAULT_JUDGE_THRESHOLD = .5;
6
+ /** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */
7
+ var SkipSignal = class extends Error {
8
+ constructor(reason) {
9
+ super("eval skipped");
10
+ this.reason = reason;
11
+ this.name = "SkipSignal";
12
+ }
13
+ };
14
+ /**
15
+ * Run a single eval against a driver. Never throws for assertion or agent
16
+ * failures — those become a non-passing {@link EvalResult}. Only a malformed
17
+ * eval definition surfaces as `result.error`.
18
+ */
19
+ async function runEval(def, options) {
20
+ const assertions = [];
21
+ let reply = "";
22
+ let lastInput = "";
23
+ let toolCalls = [];
24
+ let sessionId;
25
+ let lastTraceId;
26
+ let lastSucceeded = false;
27
+ const record = (label, pass, score, detail) => {
28
+ const result = {
29
+ label,
30
+ severity: "gate",
31
+ pass,
32
+ score,
33
+ detail
34
+ };
35
+ assertions.push(result);
36
+ const handle = {
37
+ gate() {
38
+ result.severity = "gate";
39
+ return handle;
40
+ },
41
+ soft() {
42
+ result.severity = "soft";
43
+ return handle;
44
+ },
45
+ atLeast(threshold) {
46
+ result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;
47
+ return handle;
48
+ }
49
+ };
50
+ return handle;
51
+ };
52
+ const recordJudge = (label, score, rationale) => record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);
53
+ const t = {
54
+ async send(message) {
55
+ lastInput = message;
56
+ const r = await options.driver.send(message);
57
+ reply = r.reply;
58
+ toolCalls = r.toolCalls;
59
+ sessionId = r.sessionId;
60
+ lastSucceeded = r.succeeded;
61
+ if (r.traceId) lastTraceId = r.traceId;
62
+ },
63
+ reset() {
64
+ options.driver.reset?.();
65
+ },
66
+ get reply() {
67
+ return reply;
68
+ },
69
+ get toolCalls() {
70
+ return toolCalls;
71
+ },
72
+ get sessionId() {
73
+ return sessionId;
74
+ },
75
+ get input() {
76
+ return options.row?.inputs ?? {};
77
+ },
78
+ get expected() {
79
+ return options.row?.expectations;
80
+ },
81
+ succeeded() {
82
+ return record("succeeded", lastSucceeded, void 0, lastSucceeded ? void 0 : "agent turn did not complete successfully");
83
+ },
84
+ calledTool(name) {
85
+ return record(`calledTool(${name})`, toolCalls.includes(name), void 0, `expected tool "${name}" to be called (called: ${toolCalls.length ? toolCalls.join(", ") : "none"})`);
86
+ },
87
+ check(value, matcher) {
88
+ const m = matcher(value);
89
+ return record("check", m.pass, m.score, m.detail);
90
+ },
91
+ judge: {
92
+ async factuality(expected) {
93
+ const { score, rationale } = await judgeFactuality({
94
+ input: lastInput,
95
+ output: reply,
96
+ expected
97
+ });
98
+ return recordJudge("judge.factuality", score, rationale);
99
+ },
100
+ async closedQA(criteria) {
101
+ const { score, rationale } = await judgeClosedQA({
102
+ input: lastInput,
103
+ output: reply,
104
+ criteria
105
+ });
106
+ return recordJudge("judge.closedQA", score, rationale);
107
+ },
108
+ async custom(spec) {
109
+ const { score, rationale } = await judgeCustom(spec, {
110
+ input: lastInput,
111
+ output: reply
112
+ });
113
+ return recordJudge(`judge.${spec.name}`, score, rationale);
114
+ }
115
+ },
116
+ skip(reason) {
117
+ throw new SkipSignal(reason);
118
+ }
119
+ };
120
+ try {
121
+ await def.test(t);
122
+ } catch (err) {
123
+ if (err instanceof SkipSignal) return {
124
+ id: options.id,
125
+ description: def.description,
126
+ skipped: { reason: err.reason },
127
+ assertions,
128
+ passed: true,
129
+ traceId: lastTraceId
130
+ };
131
+ return {
132
+ id: options.id,
133
+ description: def.description,
134
+ assertions,
135
+ passed: false,
136
+ error: err instanceof Error ? err.message : String(err),
137
+ traceId: lastTraceId
138
+ };
139
+ }
140
+ const passed = assertions.every((a) => a.pass || a.severity === "soft" && !options.strict);
141
+ return {
142
+ id: options.id,
143
+ description: def.description,
144
+ assertions,
145
+ passed,
146
+ traceId: lastTraceId
147
+ };
148
+ }
149
+
150
+ //#endregion
151
+ export { runEval };
152
+ //# sourceMappingURL=run-eval.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run-eval.js","names":[],"sources":["../../src/evals/run-eval.ts"],"sourcesContent":["import type { DatasetRow } from \"./dataset\";\nimport { judgeClosedQA, judgeCustom, judgeFactuality } from \"./judge\";\nimport type {\n AssertionHandle,\n AssertionResult,\n EvalDefinition,\n EvalDriver,\n EvalResult,\n Matcher,\n TestContext,\n} from \"./types\";\n\n/** Default pass threshold for an LLM-judge score (0..1) before `.atLeast()`. */\nconst DEFAULT_JUDGE_THRESHOLD = 0.5;\n\n/** Thrown by `t.skip()` to unwind the test and mark the eval skipped. */\nclass SkipSignal extends Error {\n constructor(public reason?: string) {\n super(\"eval skipped\");\n this.name = \"SkipSignal\";\n }\n}\n\nexport interface RunEvalOptions {\n /** Stable id for the eval (e.g. its file path relative to the evals dir). */\n id: string;\n /** Drives the agent and returns reply/tool-calls/success per `send`. */\n driver: EvalDriver;\n /** When true, soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Dataset row bound to `t.input`/`t.expected` for dataset-driven evals. */\n row?: DatasetRow;\n}\n\n/**\n * Run a single eval against a driver. Never throws for assertion or agent\n * failures — those become a non-passing {@link EvalResult}. Only a malformed\n * eval definition surfaces as `result.error`.\n */\nexport async function runEval(\n def: EvalDefinition,\n options: RunEvalOptions,\n): Promise<EvalResult> {\n const assertions: AssertionResult[] = [];\n let reply = \"\";\n let lastInput = \"\";\n let toolCalls: string[] = [];\n let sessionId: string | undefined;\n let lastTraceId: string | undefined;\n let lastSucceeded = false;\n\n const record = (\n label: string,\n pass: boolean,\n score?: number,\n detail?: string,\n ): AssertionHandle => {\n const result: AssertionResult = {\n label,\n severity: \"gate\",\n pass,\n score,\n detail,\n };\n assertions.push(result);\n const handle: AssertionHandle = {\n gate() {\n result.severity = \"gate\";\n return handle;\n },\n soft() {\n result.severity = \"soft\";\n return handle;\n },\n atLeast(threshold: number) {\n // Re-threshold the score; keep the current severity (gate unless the\n // caller also chained `.soft()`).\n result.pass = (result.score ?? (result.pass ? 1 : 0)) >= threshold;\n return handle;\n },\n };\n return handle;\n };\n\n // LLM-judge assertions are scored and gate by default (a miss fails the eval,\n // like other assertions). The caller chains `.atLeast(n)` to change the pass\n // threshold, or `.soft()` to demote to a tracked-only metric.\n const recordJudge = (\n label: string,\n score: number,\n rationale?: string,\n ): AssertionHandle =>\n record(label, score >= DEFAULT_JUDGE_THRESHOLD, score, rationale);\n\n const t: TestContext = {\n async send(message) {\n lastInput = message;\n const r = await options.driver.send(message);\n reply = r.reply;\n toolCalls = r.toolCalls;\n sessionId = r.sessionId;\n lastSucceeded = r.succeeded;\n if (r.traceId) lastTraceId = r.traceId;\n },\n reset() {\n options.driver.reset?.();\n },\n get reply() {\n return reply;\n },\n get toolCalls() {\n return toolCalls;\n },\n get sessionId() {\n return sessionId;\n },\n get input() {\n return options.row?.inputs ?? {};\n },\n get expected() {\n return options.row?.expectations;\n },\n succeeded() {\n return record(\n \"succeeded\",\n lastSucceeded,\n undefined,\n lastSucceeded ? undefined : \"agent turn did not complete successfully\",\n );\n },\n calledTool(name) {\n return record(\n `calledTool(${name})`,\n toolCalls.includes(name),\n undefined,\n `expected tool \"${name}\" to be called (called: ${\n toolCalls.length ? toolCalls.join(\", \") : \"none\"\n })`,\n );\n },\n check(value: string, matcher: Matcher) {\n const m = matcher(value);\n return record(\"check\", m.pass, m.score, m.detail);\n },\n judge: {\n async factuality(expected) {\n const { score, rationale } = await judgeFactuality({\n input: lastInput,\n output: reply,\n expected,\n });\n return recordJudge(\"judge.factuality\", score, rationale);\n },\n async closedQA(criteria) {\n const { score, rationale } = await judgeClosedQA({\n input: lastInput,\n output: reply,\n criteria,\n });\n return recordJudge(\"judge.closedQA\", score, rationale);\n },\n async custom(spec) {\n const { score, rationale } = await judgeCustom(spec, {\n input: lastInput,\n output: reply,\n });\n return recordJudge(`judge.${spec.name}`, score, rationale);\n },\n },\n skip(reason) {\n throw new SkipSignal(reason);\n },\n };\n\n try {\n await def.test(t);\n } catch (err) {\n if (err instanceof SkipSignal) {\n return {\n id: options.id,\n description: def.description,\n skipped: { reason: err.reason },\n assertions,\n passed: true,\n traceId: lastTraceId,\n };\n }\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n traceId: lastTraceId,\n };\n }\n\n const passed = assertions.every(\n (a) => a.pass || (a.severity === \"soft\" && !options.strict),\n );\n\n return {\n id: options.id,\n description: def.description,\n assertions,\n passed,\n traceId: lastTraceId,\n };\n}\n"],"mappings":";;;;AAaA,MAAM,0BAA0B;;AAGhC,IAAM,aAAN,cAAyB,MAAM;CAC7B,YAAY,AAAO,QAAiB;AAClC,QAAM,eAAe;EADJ;AAEjB,OAAK,OAAO;;;;;;;;AAoBhB,eAAsB,QACpB,KACA,SACqB;CACrB,MAAM,aAAgC,EAAE;CACxC,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI,YAAsB,EAAE;CAC5B,IAAI;CACJ,IAAI;CACJ,IAAI,gBAAgB;CAEpB,MAAM,UACJ,OACA,MACA,OACA,WACoB;EACpB,MAAM,SAA0B;GAC9B;GACA,UAAU;GACV;GACA;GACA;GACD;AACD,aAAW,KAAK,OAAO;EACvB,MAAM,SAA0B;GAC9B,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,OAAO;AACL,WAAO,WAAW;AAClB,WAAO;;GAET,QAAQ,WAAmB;AAGzB,WAAO,QAAQ,OAAO,UAAU,OAAO,OAAO,IAAI,OAAO;AACzD,WAAO;;GAEV;AACD,SAAO;;CAMT,MAAM,eACJ,OACA,OACA,cAEA,OAAO,OAAO,SAAS,yBAAyB,OAAO,UAAU;CAEnE,MAAM,IAAiB;EACrB,MAAM,KAAK,SAAS;AAClB,eAAY;GACZ,MAAM,IAAI,MAAM,QAAQ,OAAO,KAAK,QAAQ;AAC5C,WAAQ,EAAE;AACV,eAAY,EAAE;AACd,eAAY,EAAE;AACd,mBAAgB,EAAE;AAClB,OAAI,EAAE,QAAS,eAAc,EAAE;;EAEjC,QAAQ;AACN,WAAQ,OAAO,SAAS;;EAE1B,IAAI,QAAQ;AACV,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,YAAY;AACd,UAAO;;EAET,IAAI,QAAQ;AACV,UAAO,QAAQ,KAAK,UAAU,EAAE;;EAElC,IAAI,WAAW;AACb,UAAO,QAAQ,KAAK;;EAEtB,YAAY;AACV,UAAO,OACL,aACA,eACA,QACA,gBAAgB,SAAY,2CAC7B;;EAEH,WAAW,MAAM;AACf,UAAO,OACL,cAAc,KAAK,IACnB,UAAU,SAAS,KAAK,EACxB,QACA,kBAAkB,KAAK,0BACrB,UAAU,SAAS,UAAU,KAAK,KAAK,GAAG,OAC3C,GACF;;EAEH,MAAM,OAAe,SAAkB;GACrC,MAAM,IAAI,QAAQ,MAAM;AACxB,UAAO,OAAO,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,OAAO;;EAEnD,OAAO;GACL,MAAM,WAAW,UAAU;IACzB,MAAM,EAAE,OAAO,cAAc,MAAM,gBAAgB;KACjD,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,oBAAoB,OAAO,UAAU;;GAE1D,MAAM,SAAS,UAAU;IACvB,MAAM,EAAE,OAAO,cAAc,MAAM,cAAc;KAC/C,OAAO;KACP,QAAQ;KACR;KACD,CAAC;AACF,WAAO,YAAY,kBAAkB,OAAO,UAAU;;GAExD,MAAM,OAAO,MAAM;IACjB,MAAM,EAAE,OAAO,cAAc,MAAM,YAAY,MAAM;KACnD,OAAO;KACP,QAAQ;KACT,CAAC;AACF,WAAO,YAAY,SAAS,KAAK,QAAQ,OAAO,UAAU;;GAE7D;EACD,KAAK,QAAQ;AACX,SAAM,IAAI,WAAW,OAAO;;EAE/B;AAED,KAAI;AACF,QAAM,IAAI,KAAK,EAAE;UACV,KAAK;AACZ,MAAI,eAAe,WACjB,QAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB,SAAS,EAAE,QAAQ,IAAI,QAAQ;GAC/B;GACA,QAAQ;GACR,SAAS;GACV;AAEH,SAAO;GACL,IAAI,QAAQ;GACZ,aAAa,IAAI;GACjB;GACA,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACvD,SAAS;GACV;;CAGH,MAAM,SAAS,WAAW,OACvB,MAAM,EAAE,QAAS,EAAE,aAAa,UAAU,CAAC,QAAQ,OACrD;AAED,QAAO;EACL,IAAI,QAAQ;EACZ,aAAa,IAAI;EACjB;EACA;EACA,SAAS;EACV"}
@@ -0,0 +1,94 @@
1
+ import { WorkspaceClient } from "../workspace-client/index.js";
2
+ import "./dataset.js";
3
+ import { EvalResult } from "./types.js";
4
+ import { ReportOutcome } from "./mlflow-report.js";
5
+ import { FinishOutcome } from "./mlflow-run.js";
6
+
7
+ //#region src/evals/run-evals.d.ts
8
+ interface RunEvalsOptions {
9
+ /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */
10
+ rootDir?: string;
11
+ /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */
12
+ baseUrl: string;
13
+ /** Substring filter on `<agent>/<id>` (or an exact agent id). */
14
+ filter?: string;
15
+ /** Soft assertion failures also fail the eval. */
16
+ strict?: boolean;
17
+ /** Extra request headers for the driver (e.g. auth for a deployed app). */
18
+ headers?: Record<string, string>;
19
+ /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */
20
+ timeoutMs?: number;
21
+ /**
22
+ * Max evals to drive concurrently. Each eval opens one stream to the app as
23
+ * the same user, so keep this at or below the app's
24
+ * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the
25
+ * 429 guard. Defaults to 4; clamped to `[1, total]`.
26
+ */
27
+ concurrency?: number;
28
+ /**
29
+ * When set, create a native MLflow "Evaluation run": each eval's trace is
30
+ * linked to the run, pass/fail is written as feedback, and aggregate metrics
31
+ * are logged. Requires Databricks creds + the target experiment.
32
+ */
33
+ mlflow?: {
34
+ host: string;
35
+ token: string;
36
+ experimentId: string; /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */
37
+ sqlWarehouseId?: string;
38
+ };
39
+ /**
40
+ * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a
41
+ * Databricks serving endpoint (`model`).
42
+ */
43
+ judge?: {
44
+ host: string;
45
+ token: string;
46
+ model: string;
47
+ };
48
+ /**
49
+ * Workspace client used to read managed evaluation datasets (for evals that
50
+ * declare `dataset`). Required alongside {@link warehouseId} for those evals.
51
+ */
52
+ workspaceClient?: WorkspaceClient;
53
+ /** SQL warehouse id used to read managed evaluation datasets. */
54
+ warehouseId?: string;
55
+ /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */
56
+ now?: number;
57
+ /** Progress callback, invoked as evals are discovered, started, and finished. */
58
+ onEvent?: (event: EvalProgress) => void;
59
+ }
60
+ type EvalProgress = {
61
+ type: "discovered";
62
+ total: number;
63
+ } | {
64
+ type: "run-created";
65
+ runId: string;
66
+ } | {
67
+ type: "start";
68
+ id: string;
69
+ index: number;
70
+ total: number;
71
+ } | {
72
+ type: "result";
73
+ result: EvalResult;
74
+ index: number;
75
+ total: number;
76
+ };
77
+ interface EvalRunSummary {
78
+ results: EvalResult[];
79
+ /** Present when an MLflow evaluation run was created. */
80
+ mlflow?: {
81
+ runId: string;
82
+ report: ReportOutcome;
83
+ finish: FinishOutcome;
84
+ };
85
+ }
86
+ /**
87
+ * Discover, load, and run every eval under each agent's `evals/` dir, driving
88
+ * the agents on a running app. Never throws for an individual eval — load/run
89
+ * failures become non-passing {@link EvalResult}s.
90
+ */
91
+ declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
92
+ //#endregion
93
+ export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir };
94
+ //# sourceMappingURL=run-evals.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAciB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAMU;EAJV,MAAA;EAyCkB;EAvClB,MAAA;EAuC8B;EArC9B,OAAA,GAAU,MAAA;EANV;EAQA,SAAA;EAJA;;;;;;EAWA,WAAA;EAQE;;;;;EAFF,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UAeF;IAbE,cAAA;EAAA;EAiBgB;;;;EAXlB,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EAef;;;;EAVxB,eAAA,GAAkB,eAAA;EAYa;EAV/B,WAAA;EAWI;EATJ,GAAA;EAS4B;EAP5B,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EAE6D;EAAtE,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;iBAoOrC,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
@@ -0,0 +1,257 @@
1
+ import { MlflowClient } from "../connectors/mlflow/client.js";
2
+ import { readEvalDataset } from "./dataset.js";
3
+ import { discoverEvalFiles } from "./discover.js";
4
+ import { createHttpDriver } from "./http-driver.js";
5
+ import { configureJudge, teardownJudge } from "./judge.js";
6
+ import { mapPool } from "./pool.js";
7
+ import { reportToMlflow } from "./mlflow-report.js";
8
+ import { runEval } from "./run-eval.js";
9
+ import { createEvalRun, finishEvalRun } from "./mlflow-run.js";
10
+ import { pathToFileURL } from "node:url";
11
+
12
+ //#region src/evals/run-evals.ts
13
+ /**
14
+ * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.
15
+ * Uses tsx's programmatic loader so TypeScript eval files run without a build
16
+ * step. The specifier is indirected so the type checker doesn't try to resolve
17
+ * tsx's internal entry.
18
+ */
19
+ async function loadEval(file) {
20
+ const tsxApi = "tsx/esm/api";
21
+ let tsImport;
22
+ try {
23
+ ({tsImport} = await import(tsxApi));
24
+ } catch {
25
+ throw new Error("Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).");
26
+ }
27
+ const def = resolveEvalDefault(await tsImport(pathToFileURL(file).href, import.meta.url));
28
+ if (!def) throw new Error(`${file}: must default-export defineEval({ test })`);
29
+ return def;
30
+ }
31
+ /**
32
+ * Unwrap the eval default export across module-interop shapes. Depending on
33
+ * whether the eval file is treated as ESM or CJS, the value lands at
34
+ * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or
35
+ * `mod` itself. Returns the first candidate that looks like an eval.
36
+ */
37
+ function resolveEvalDefault(mod) {
38
+ let candidate = mod;
39
+ for (let i = 0; i < 4 && candidate; i++) {
40
+ if (typeof candidate.test === "function") return candidate;
41
+ candidate = candidate.default;
42
+ }
43
+ }
44
+ /**
45
+ * Run one eval turn against a fresh driver. Never throws — a run failure becomes
46
+ * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`
47
+ * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is
48
+ * `undefined` for a plain single-run eval.
49
+ */
50
+ async function runOne(d, id, def, row, runId, options) {
51
+ try {
52
+ return await runEval(def, {
53
+ id,
54
+ driver: createHttpDriver({
55
+ baseUrl: options.baseUrl,
56
+ agent: def.agent ?? d.agent,
57
+ headers: options.headers,
58
+ mlflowRunId: runId,
59
+ timeoutMs: options.timeoutMs
60
+ }),
61
+ strict: options.strict,
62
+ row
63
+ });
64
+ } catch (err) {
65
+ return {
66
+ id,
67
+ assertions: [],
68
+ passed: false,
69
+ error: err instanceof Error ? err.message : String(err)
70
+ };
71
+ }
72
+ }
73
+ /**
74
+ * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval
75
+ * yields a single `undefined` row; a dataset eval reads its Unity Catalog table
76
+ * via {@link readEvalDataset}. On misconfiguration or read failure, returns a
77
+ * single `undefined` row plus an `error`, so the eval still surfaces one result.
78
+ */
79
+ async function resolveDatasetRows(def, options) {
80
+ if (!def.dataset) return { rows: [void 0] };
81
+ if (!options.workspaceClient || !options.warehouseId) return {
82
+ rows: [void 0],
83
+ error: "dataset eval requires a workspace client and warehouse (pass --warehouse-id)"
84
+ };
85
+ try {
86
+ const rows = await readEvalDataset(options.workspaceClient, {
87
+ table: def.dataset.table,
88
+ warehouseId: options.warehouseId,
89
+ limit: def.dataset.limit
90
+ });
91
+ if (rows.length === 0) return {
92
+ rows: [void 0],
93
+ error: `dataset "${def.dataset.table}" returned no rows`
94
+ };
95
+ return { rows };
96
+ } catch (err) {
97
+ return {
98
+ rows: [void 0],
99
+ error: err instanceof Error ? err.message : String(err)
100
+ };
101
+ }
102
+ }
103
+ /**
104
+ * Load one discovered eval and run it, expanding a dataset-driven eval into one
105
+ * run per row. Appends one result per row to `results`, emitting `start`/
106
+ * `result` around each. Never throws: a load or dataset-read failure surfaces as
107
+ * a non-passing result. `total` counts eval files, not rows — per-row detail is
108
+ * carried in the result id (`[row i/n]`).
109
+ */
110
+ async function runDiscovered(d, index, total, runId, options, emit, results) {
111
+ const id = `${d.agent}/${d.id}`;
112
+ let def;
113
+ try {
114
+ def = await loadEval(d.file);
115
+ } catch (err) {
116
+ emit({
117
+ type: "start",
118
+ id,
119
+ index,
120
+ total
121
+ });
122
+ const result = {
123
+ id,
124
+ assertions: [],
125
+ passed: false,
126
+ error: err instanceof Error ? err.message : String(err)
127
+ };
128
+ results.push(result);
129
+ emit({
130
+ type: "result",
131
+ result,
132
+ index,
133
+ total
134
+ });
135
+ return;
136
+ }
137
+ const { rows, error: datasetError } = await resolveDatasetRows(def, options);
138
+ for (let r = 0; r < rows.length; r++) {
139
+ const rowId = def.dataset && rows.length > 1 ? `${id} [row ${r + 1}/${rows.length}]` : id;
140
+ emit({
141
+ type: "start",
142
+ id: rowId,
143
+ index,
144
+ total
145
+ });
146
+ const result = datasetError ? {
147
+ id: rowId,
148
+ assertions: [],
149
+ passed: false,
150
+ error: datasetError
151
+ } : await runOne(d, rowId, def, rows[r], runId, options);
152
+ results.push(result);
153
+ emit({
154
+ type: "result",
155
+ result,
156
+ index,
157
+ total
158
+ });
159
+ }
160
+ }
161
+ /** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */
162
+ async function maybeConfigureJudge(options) {
163
+ if (!options.judge) return;
164
+ await configureJudge({
165
+ client: new MlflowClient(options.judge.host, options.judge.token),
166
+ token: options.judge.token,
167
+ model: options.judge.model
168
+ });
169
+ }
170
+ /**
171
+ * Report per-eval assessments and finish the MLflow run, when one was created.
172
+ * Returns the run summary, or `undefined` when there was no run to finalize.
173
+ */
174
+ async function finalizeMlflow(client, runId, results, options) {
175
+ if (!client || !runId) return void 0;
176
+ let report = {
177
+ written: 0,
178
+ skipped: 0,
179
+ failures: []
180
+ };
181
+ try {
182
+ report = await reportToMlflow(client, results, options.mlflow?.sqlWarehouseId);
183
+ } catch (err) {
184
+ report.failures.push({
185
+ traceId: "(report)",
186
+ error: err instanceof Error ? err.message : String(err)
187
+ });
188
+ }
189
+ const finish = await finishEvalRun(client, {
190
+ runId,
191
+ results,
192
+ endTime: options.now ?? Date.now()
193
+ });
194
+ return {
195
+ runId,
196
+ report,
197
+ finish
198
+ };
199
+ }
200
+ /**
201
+ * Default max evals in flight. Each eval opens one stream to the app as the
202
+ * same user; the server caps concurrent streams per user at 5 by default
203
+ * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.
204
+ */
205
+ const DEFAULT_CONCURRENCY = 4;
206
+ /**
207
+ * Discover, load, and run every eval under each agent's `evals/` dir, driving
208
+ * the agents on a running app. Never throws for an individual eval — load/run
209
+ * failures become non-passing {@link EvalResult}s.
210
+ */
211
+ async function runEvalsInDir(options) {
212
+ const root = options.rootDir ?? process.cwd();
213
+ const now = options.now ?? Date.now();
214
+ let discovered = discoverEvalFiles(root);
215
+ if (options.filter) {
216
+ const f = options.filter;
217
+ discovered = discovered.filter((d) => d.agent === f || `${d.agent}/${d.id}`.includes(f));
218
+ }
219
+ const emit = options.onEvent ?? (() => {});
220
+ const total = discovered.length;
221
+ emit({
222
+ type: "discovered",
223
+ total
224
+ });
225
+ await maybeConfigureJudge(options);
226
+ try {
227
+ let runId;
228
+ let mlflowClient;
229
+ if (options.mlflow) {
230
+ mlflowClient = new MlflowClient(options.mlflow.host, options.mlflow.token);
231
+ runId = await createEvalRun(mlflowClient, {
232
+ experimentId: options.mlflow.experimentId,
233
+ runName: `appkit-eval ${new Date(now).toISOString()}`,
234
+ startTime: now
235
+ });
236
+ emit({
237
+ type: "run-created",
238
+ runId
239
+ });
240
+ }
241
+ const results = (await mapPool(discovered, options.concurrency ?? DEFAULT_CONCURRENCY, async (d, index) => {
242
+ const fileResults = [];
243
+ await runDiscovered(d, index, total, runId, options, emit, fileResults);
244
+ return fileResults;
245
+ })).flat();
246
+ const summary = { results };
247
+ const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);
248
+ if (mlflow) summary.mlflow = mlflow;
249
+ return summary;
250
+ } finally {
251
+ teardownJudge();
252
+ }
253
+ }
254
+
255
+ //#endregion
256
+ export { runEvalsInDir };
257
+ //# sourceMappingURL=run-evals.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run-evals.js","names":[],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport { type DiscoveredEval, discoverEvalFiles } from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */\n timeoutMs?: number;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n * Uses tsx's programmatic loader so TypeScript eval files run without a build\n * step. The specifier is indirected so the type checker doesn't try to resolve\n * tsx's internal entry.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n\n const mod = await tsImport(pathToFileURL(file).href, import.meta.url);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // A fresh driver per row: each row is an independent conversation whose\n // thread must not carry over the previous row's history. (Multiple\n // `t.send`s within one row still share the thread — the driver's behavior.)\n const driver = createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n timeoutMs: options.timeoutMs,\n });\n return await runEval(def, { id, driver, strict: options.strict, row });\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Load one discovered eval and run it, expanding a dataset-driven eval into one\n * run per row. Appends one result per row to `results`, emitting `start`/\n * `result` around each. Never throws: a load or dataset-read failure surfaces as\n * a non-passing result. `total` counts eval files, not rows — per-row detail is\n * carried in the result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n const total = discovered.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each eval through the bounded pool — one in-flight stream per eval, so\n // the pool respects the server's per-user stream cap (see mapPool/concurrency).\n // A dataset eval expands into per-row runs that execute serially within its\n // slot; results preserve discovery order (mapPool writes by index) and row\n // order within each file. `total` counts eval files, not dataset rows — per-row\n // detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n discovered,\n options.concurrency ?? DEFAULT_CONCURRENCY,\n async (d, index) => {\n const fileResults: EvalResult[] = [];\n await runDiscovered(d, index, total, runId, options, emit, fileResults);\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAkFA,eAAe,SAAS,MAAuC;CAC7D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;CAIH,MAAM,MAAM,mBADA,MAAM,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI,CAClC;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;;AAST,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAWF,SAAO,MAAM,QAAQ,KAAK;GAAE;GAAI,QAPjB,iBAAiB;IAC9B,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IACb,WAAW,QAAQ;IACpB,CAAC;GACsC,QAAQ,QAAQ;GAAQ;GAAK,CAAC;UAC/D,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;AAWL,eAAe,cACb,GACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;CAE3B,IAAI;AACJ,KAAI;AACF,QAAM,MAAM,SAAS,EAAE,KAAK;UACrB,KAAK;AACZ,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;AAO5B,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CACvC,MAAM,QAAQ,WAAW;AACzB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkBtC,MAAM,WATU,MAAM,QACpB,YACA,QAAQ,eAAe,qBACvB,OAAO,GAAG,UAAU;GAClB,MAAM,cAA4B,EAAE;AACpC,SAAM,cAAc,GAAG,OAAO,OAAO,OAAO,SAAS,MAAM,YAAY;AACvE,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}