@databricks/appkit 0.73.0 → 0.74.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/CLAUDE.md +7 -0
  2. package/dist/appkit/package.js +1 -1
  3. package/dist/beta.d.ts +6 -6
  4. package/dist/beta.js +6 -6
  5. package/dist/cli/commands/agent/eval.js +85 -21
  6. package/dist/cli/commands/agent/eval.js.map +1 -1
  7. package/dist/cli/commands/registry/add.js +77 -6
  8. package/dist/cli/commands/registry/add.js.map +1 -1
  9. package/dist/evals/dataset.d.ts +14 -1
  10. package/dist/evals/dataset.d.ts.map +1 -1
  11. package/dist/evals/dataset.js +16 -1
  12. package/dist/evals/dataset.js.map +1 -1
  13. package/dist/evals/define-eval.d.ts +4 -2
  14. package/dist/evals/define-eval.d.ts.map +1 -1
  15. package/dist/evals/define-eval.js +5 -1
  16. package/dist/evals/define-eval.js.map +1 -1
  17. package/dist/evals/discover.d.ts +15 -1
  18. package/dist/evals/discover.d.ts.map +1 -1
  19. package/dist/evals/discover.js +40 -10
  20. package/dist/evals/discover.js.map +1 -1
  21. package/dist/evals/http-driver.d.ts.map +1 -1
  22. package/dist/evals/http-driver.js +31 -9
  23. package/dist/evals/http-driver.js.map +1 -1
  24. package/dist/evals/index.d.ts +5 -5
  25. package/dist/evals/index.js +5 -5
  26. package/dist/evals/report.d.ts +16 -1
  27. package/dist/evals/report.d.ts.map +1 -1
  28. package/dist/evals/report.js +64 -2
  29. package/dist/evals/report.js.map +1 -1
  30. package/dist/evals/run-eval.d.ts +5 -0
  31. package/dist/evals/run-eval.d.ts.map +1 -1
  32. package/dist/evals/run-eval.js +44 -3
  33. package/dist/evals/run-eval.js.map +1 -1
  34. package/dist/evals/run-evals.d.ts +32 -3
  35. package/dist/evals/run-evals.d.ts.map +1 -1
  36. package/dist/evals/run-evals.js +128 -26
  37. package/dist/evals/run-evals.js.map +1 -1
  38. package/dist/evals/types.d.ts +40 -2
  39. package/dist/evals/types.d.ts.map +1 -1
  40. package/dist/registry/manifest-loader.d.ts +1 -1
  41. package/docs/api/appkit/Function.defineEvalConfig.md +18 -0
  42. package/docs/api/appkit/Function.discoverEvalConfigs.md +18 -0
  43. package/docs/api/appkit/Function.formatResultsJUnit.md +18 -0
  44. package/docs/api/appkit/Function.formatResultsJson.md +18 -0
  45. package/docs/api/appkit/Function.runWithRetries.md +28 -0
  46. package/docs/api/appkit/Function.userTurns.md +20 -0
  47. package/docs/api/appkit/Interface.DiscoveredEvalConfig.md +25 -0
  48. package/docs/api/appkit/Interface.DriveResult.md +28 -0
  49. package/docs/api/appkit/Interface.EvalDefinition.md +22 -0
  50. package/docs/api/appkit/Interface.EvalDriver.md +10 -4
  51. package/docs/api/appkit/Interface.EvalResult.md +11 -0
  52. package/docs/api/appkit/Interface.EvalSummary.md +11 -0
  53. package/docs/api/appkit/Interface.RunEvalOptions.md +11 -0
  54. package/docs/api/appkit/Interface.RunEvalsOptions.md +23 -1
  55. package/docs/api/appkit/Interface.TestContext.md +30 -8
  56. package/docs/api/appkit.md +7 -0
  57. package/docs/plugins/agents.md +255 -4
  58. package/llms.txt +7 -0
  59. package/package.json +1 -1
  60. package/sbom.cdx.json +1 -1
@@ -1,6 +1,6 @@
1
1
  import { MlflowClient } from "../connectors/mlflow/client.js";
2
2
  import { readEvalDataset } from "./dataset.js";
3
- import { discoverEvalFiles } from "./discover.js";
3
+ import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
4
4
  import { createHttpDriver } from "./http-driver.js";
5
5
  import { configureJudge, teardownJudge } from "./judge.js";
6
6
  import { mapPool } from "./pool.js";
@@ -8,15 +8,15 @@ import { reportToMlflow } from "./mlflow-report.js";
8
8
  import { runEval } from "./run-eval.js";
9
9
  import { createEvalRun, finishEvalRun } from "./mlflow-run.js";
10
10
  import { pathToFileURL } from "node:url";
11
+ import { setTimeout } from "node:timers/promises";
11
12
 
12
13
  //#region src/evals/run-evals.ts
13
14
  /**
14
- * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.
15
- * Uses tsx's programmatic loader so TypeScript eval files run without a build
16
- * step. The specifier is indirected so the type checker doesn't try to resolve
17
- * tsx's internal entry.
15
+ * Import a TypeScript file with tsx's programmatic loader so eval files run
16
+ * without a build step. The specifier is indirected so the type checker doesn't
17
+ * try to resolve tsx's internal entry.
18
18
  */
19
- async function loadEval(file) {
19
+ async function tsImportFile(file) {
20
20
  const tsxApi = "tsx/esm/api";
21
21
  let tsImport;
22
22
  try {
@@ -24,11 +24,38 @@ async function loadEval(file) {
24
24
  } catch {
25
25
  throw new Error("Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).");
26
26
  }
27
- const def = resolveEvalDefault(await tsImport(pathToFileURL(file).href, import.meta.url));
27
+ return tsImport(pathToFileURL(file).href, import.meta.url);
28
+ }
29
+ /**
30
+ * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.
31
+ */
32
+ async function loadEval(file) {
33
+ const def = resolveEvalDefault(await tsImportFile(file));
28
34
  if (!def) throw new Error(`${file}: must default-export defineEval({ test })`);
29
35
  return def;
30
36
  }
31
37
  /**
38
+ * Load an `evals.config.ts` file and return its default-exported
39
+ * {@link EvalConfig}. A malformed/missing default surfaces as `undefined` so a
40
+ * bad config never aborts a whole run.
41
+ */
42
+ async function loadEvalConfig(file) {
43
+ return resolveConfigDefault(await tsImportFile(file));
44
+ }
45
+ /**
46
+ * Unwrap the config default export across module-interop shapes (see
47
+ * {@link resolveEvalDefault}). A config has no `.test`, so the first plain
48
+ * object reached through the `default` chain is taken as the config.
49
+ */
50
+ function resolveConfigDefault(mod) {
51
+ let candidate = mod;
52
+ for (let i = 0; i < 4 && candidate; i++) {
53
+ const next = candidate.default;
54
+ if (next === void 0) return typeof candidate === "object" ? candidate : void 0;
55
+ candidate = next;
56
+ }
57
+ }
58
+ /**
32
59
  * Unwrap the eval default export across module-interop shapes. Depending on
33
60
  * whether the eval file is treated as ESM or CJS, the value lands at
34
61
  * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or
@@ -49,18 +76,19 @@ function resolveEvalDefault(mod) {
49
76
  */
50
77
  async function runOne(d, id, def, row, runId, options) {
51
78
  try {
52
- return await runEval(def, {
79
+ return await runWithRetries(options.retries ?? 0, () => runEval(def, {
53
80
  id,
54
81
  driver: createHttpDriver({
55
82
  baseUrl: options.baseUrl,
56
83
  agent: def.agent ?? d.agent,
57
84
  headers: options.headers,
58
85
  mlflowRunId: runId,
59
- timeoutMs: options.timeoutMs
86
+ timeoutMs: def.timeoutMs ?? options.timeoutMs
60
87
  }),
61
88
  strict: options.strict,
62
- row
63
- });
89
+ row,
90
+ timeoutMs: options.timeoutMs
91
+ }));
64
92
  } catch (err) {
65
93
  return {
66
94
  id,
@@ -101,18 +129,16 @@ async function resolveDatasetRows(def, options) {
101
129
  }
102
130
  }
103
131
  /**
104
- * Load one discovered eval and run it, expanding a dataset-driven eval into one
105
- * run per row. Appends one result per row to `results`, emitting `start`/
106
- * `result` around each. Never throws: a load or dataset-read failure surfaces as
107
- * a non-passing result. `total` counts eval files, not rows — per-row detail is
108
- * carried in the result id (`[row i/n]`).
132
+ * Run one already-loaded eval (from the `loaded` pre-pass), expanding a
133
+ * dataset-driven eval into one run per row. Appends one result per row to
134
+ * `results`, emitting `start`/`result` around each. Never throws: a load error
135
+ * (carried in `loadError`) or a dataset-read failure surfaces as a non-passing
136
+ * result. `total` counts eval files, not rows — per-row detail is carried in the
137
+ * result id (`[row i/n]`).
109
138
  */
110
- async function runDiscovered(d, index, total, runId, options, emit, results) {
139
+ async function runDiscovered(d, def, loadError, index, total, runId, options, emit, results) {
111
140
  const id = `${d.agent}/${d.id}`;
112
- let def;
113
- try {
114
- def = await loadEval(d.file);
115
- } catch (err) {
141
+ if (loadError) {
116
142
  emit({
117
143
  type: "start",
118
144
  id,
@@ -123,7 +149,7 @@ async function runDiscovered(d, index, total, runId, options, emit, results) {
123
149
  id,
124
150
  assertions: [],
125
151
  passed: false,
126
- error: err instanceof Error ? err.message : String(err)
152
+ error: loadError
127
153
  };
128
154
  results.push(result);
129
155
  emit({
@@ -158,6 +184,43 @@ async function runDiscovered(d, index, total, runId, options, emit, results) {
158
184
  });
159
185
  }
160
186
  }
187
+ /** Base delay (ms) before the first retry; doubled per attempt, full-jittered, capped. */
188
+ const DEFAULT_RETRY_BASE_DELAY_MS = 250;
189
+ /** Ceiling for a single retry backoff wait (ms). */
190
+ const MAX_RETRY_DELAY_MS = 5e3;
191
+ /**
192
+ * Run `attempt` up to `1 + retries` times, stopping as soon as it returns a
193
+ * result that is neither a thrown error / per-eval timeout (`error`) nor a
194
+ * transport/agent turn failure (`infraFailure`). Assertion failures set
195
+ * neither, so a failed-but-completed eval is returned on the first try and
196
+ * never retried. Returns the last result when every attempt failed on infra.
197
+ *
198
+ * Between attempts it waits a full-jittered exponential backoff (infra flakes
199
+ * are overload-correlated). `retries` is coerced to a finite non-negative
200
+ * integer; `baseDelayMs: 0` disables the wait (tests).
201
+ */
202
+ async function runWithRetries(retries, attempt, options = {}) {
203
+ const baseDelayMs = options.baseDelayMs ?? DEFAULT_RETRY_BASE_DELAY_MS;
204
+ const maxAttempts = 1 + (Number.isFinite(retries) ? Math.max(0, Math.floor(retries)) : 0);
205
+ let result;
206
+ for (let n = 1;; n++) {
207
+ result = await attempt(n);
208
+ if (!(result.error !== void 0 || result.infraFailure) || n >= maxAttempts) return result;
209
+ if (baseDelayMs > 0) {
210
+ const ceiling = Math.min(baseDelayMs * 2 ** (n - 1), MAX_RETRY_DELAY_MS);
211
+ await setTimeout(Math.random() * ceiling);
212
+ }
213
+ }
214
+ }
215
+ /**
216
+ * Whether an eval's `tags` satisfy a `--tag` filter: `true` when the filter is
217
+ * empty/undefined (no filtering), otherwise only when the eval shares at least
218
+ * one tag with it. An eval with no tags never matches a non-empty filter.
219
+ */
220
+ function matchesTags(defTags, filterTags) {
221
+ if (!filterTags || filterTags.length === 0) return true;
222
+ return defTags?.some((t) => filterTags.includes(t)) ?? false;
223
+ }
161
224
  /** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */
162
225
  async function maybeConfigureJudge(options) {
163
226
  if (!options.judge) return;
@@ -204,6 +267,16 @@ async function finalizeMlflow(client, runId, results, options) {
204
267
  */
205
268
  const DEFAULT_CONCURRENCY = 4;
206
269
  /**
270
+ * Resolve the work-pool width: `--concurrency` wins; else the lowest
271
+ * `maxConcurrency` any *participating* agent's `evals.config.ts` requests (all
272
+ * evals share one per-user stream budget, so the most conservative ceiling
273
+ * governs); else {@link DEFAULT_CONCURRENCY}.
274
+ */
275
+ function deriveConcurrency(activeAgents, configs, cliConcurrency) {
276
+ const configMin = [...configs.entries()].filter(([agent]) => activeAgents.has(agent)).map(([, c]) => c.maxConcurrency).filter((n) => typeof n === "number").reduce((min, n) => min === void 0 ? n : Math.min(min, n), void 0);
277
+ return cliConcurrency ?? configMin ?? DEFAULT_CONCURRENCY;
278
+ }
279
+ /**
207
280
  * Discover, load, and run every eval under each agent's `evals/` dir, driving
208
281
  * the agents on a running app. Never throws for an individual eval — load/run
209
282
  * failures become non-passing {@link EvalResult}s.
@@ -217,7 +290,32 @@ async function runEvalsInDir(options) {
217
290
  discovered = discovered.filter((d) => d.agent === f || `${d.agent}/${d.id}`.includes(f));
218
291
  }
219
292
  const emit = options.onEvent ?? (() => {});
220
- const total = discovered.length;
293
+ const configs = /* @__PURE__ */ new Map();
294
+ for (const c of discoverEvalConfigs(root)) try {
295
+ const cfg = await loadEvalConfig(c.file);
296
+ if (cfg) configs.set(c.agent, cfg);
297
+ } catch {}
298
+ const loaded = [];
299
+ for (const d of discovered) {
300
+ let def;
301
+ try {
302
+ def = await loadEval(d.file);
303
+ } catch (err) {
304
+ loaded.push({
305
+ d,
306
+ def: { test: () => {} },
307
+ loadError: err instanceof Error ? err.message : String(err)
308
+ });
309
+ continue;
310
+ }
311
+ if (!matchesTags(def.tags, options.tags)) continue;
312
+ loaded.push({
313
+ d,
314
+ def
315
+ });
316
+ }
317
+ const concurrency = deriveConcurrency(new Set(loaded.map((l) => l.d.agent)), configs, options.concurrency);
318
+ const total = loaded.length;
221
319
  emit({
222
320
  type: "discovered",
223
321
  total
@@ -238,9 +336,13 @@ async function runEvalsInDir(options) {
238
336
  runId
239
337
  });
240
338
  }
241
- const results = (await mapPool(discovered, options.concurrency ?? DEFAULT_CONCURRENCY, async (d, index) => {
339
+ const results = (await mapPool(loaded, concurrency, async ({ d, def, loadError }, index) => {
242
340
  const fileResults = [];
243
- await runDiscovered(d, index, total, runId, options, emit, fileResults);
341
+ const fileOptions = {
342
+ ...options,
343
+ timeoutMs: options.timeoutMs ?? configs.get(d.agent)?.timeoutMs
344
+ };
345
+ await runDiscovered(d, def, loadError, index, total, runId, fileOptions, emit, fileResults);
244
346
  return fileResults;
245
347
  })).flat();
246
348
  const summary = { results };
@@ -253,5 +355,5 @@ async function runEvalsInDir(options) {
253
355
  }
254
356
 
255
357
  //#endregion
256
- export { runEvalsInDir };
358
+ export { runEvalsInDir, runWithRetries };
257
359
  //# sourceMappingURL=run-evals.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"run-evals.js","names":[],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport { type DiscoveredEval, discoverEvalFiles } from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /** Per-turn wall-clock timeout (ms) before a turn is failed. Defaults to 120s. */\n timeoutMs?: number;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n * Uses tsx's programmatic loader so TypeScript eval files run without a build\n * step. The specifier is indirected so the type checker doesn't try to resolve\n * tsx's internal entry.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n\n const mod = await tsImport(pathToFileURL(file).href, import.meta.url);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // A fresh driver per row: each row is an independent conversation whose\n // thread must not carry over the previous row's history. (Multiple\n // `t.send`s within one row still share the thread — the driver's behavior.)\n const driver = createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n timeoutMs: options.timeoutMs,\n });\n return await runEval(def, { id, driver, strict: options.strict, row });\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Load one discovered eval and run it, expanding a dataset-driven eval into one\n * run per row. Appends one result per row to `results`, emitting `start`/\n * `result` around each. Never throws: a load or dataset-read failure surfaces as\n * a non-passing result. `total` counts eval files, not rows — per-row detail is\n * carried in the result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n const total = discovered.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each eval through the bounded pool — one in-flight stream per eval, so\n // the pool respects the server's per-user stream cap (see mapPool/concurrency).\n // A dataset eval expands into per-row runs that execute serially within its\n // slot; results preserve discovery order (mapPool writes by index) and row\n // order within each file. `total` counts eval files, not dataset rows — per-row\n // detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n discovered,\n options.concurrency ?? DEFAULT_CONCURRENCY,\n async (d, index) => {\n const fileResults: EvalResult[] = [];\n await runDiscovered(d, index, total, runId, options, emit, fileResults);\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAkFA,eAAe,SAAS,MAAuC;CAC7D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;CAIH,MAAM,MAAM,mBADA,MAAM,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI,CAClC;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;;AAST,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAWF,SAAO,MAAM,QAAQ,KAAK;GAAE;GAAI,QAPjB,iBAAiB;IAC9B,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IACb,WAAW,QAAQ;IACpB,CAAC;GACsC,QAAQ,QAAQ;GAAQ;GAAK,CAAC;UAC/D,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;AAWL,eAAe,cACb,GACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;CAE3B,IAAI;AACJ,KAAI;AACF,QAAM,MAAM,SAAS,EAAE,KAAK;UACrB,KAAK;AACZ,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;AAO5B,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CACvC,MAAM,QAAQ,WAAW;AACzB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkBtC,MAAM,WATU,MAAM,QACpB,YACA,QAAQ,eAAe,qBACvB,OAAO,GAAG,UAAU;GAClB,MAAM,cAA4B,EAAE;AACpC,SAAM,cAAc,GAAG,OAAO,OAAO,OAAO,SAAS,MAAM,YAAY;AACvE,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
1
+ {"version":3,"file":"run-evals.js","names":["sleep"],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { setTimeout as sleep } from \"node:timers/promises\";\nimport { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport {\n type DiscoveredEval,\n discoverEvalConfigs,\n discoverEvalFiles,\n} from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalConfig, EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /**\n * Only run evals whose `tags` intersect this list. Empty/undefined runs all.\n * Tags live on the eval def, so filtering happens after each file is loaded.\n */\n tags?: string[];\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /**\n * Default per-eval timeout (ms): `runEval` races the whole test against it and\n * it also caps each driver turn. A per-eval `def.timeoutMs` overrides it, and\n * it wins over an agent's `evals.config.ts` `timeoutMs`. Unbounded when unset.\n */\n timeoutMs?: number;\n /**\n * Re-run an eval up to this many extra times when it fails on infrastructure —\n * a thrown error/timeout (`result.error`) or a transport/agent turn failure\n * (`result.infraFailure`). Assertion failures are never retried. Defaults to `0`.\n */\n retries?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Import a TypeScript file with tsx's programmatic loader so eval files run\n * without a build step. The specifier is indirected so the type checker doesn't\n * try to resolve tsx's internal entry.\n */\nasync function tsImportFile(file: string): Promise<unknown> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n return tsImport(pathToFileURL(file).href, import.meta.url);\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const mod = await tsImportFile(file);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Load an `evals.config.ts` file and return its default-exported\n * {@link EvalConfig}. A malformed/missing default surfaces as `undefined` so a\n * bad config never aborts a whole run.\n */\nasync function loadEvalConfig(file: string): Promise<EvalConfig | undefined> {\n const mod = await tsImportFile(file);\n return resolveConfigDefault(mod);\n}\n\n/**\n * Unwrap the config default export across module-interop shapes (see\n * {@link resolveEvalDefault}). A config has no `.test`, so the first plain\n * object reached through the `default` chain is taken as the config.\n */\nexport function resolveConfigDefault(mod: unknown): EvalConfig | undefined {\n let candidate: unknown = mod;\n // `i < 4` bounds the chain; no visited-set needed (cf. resolveEvalDefault).\n for (let i = 0; i < 4 && candidate; i++) {\n const next = (candidate as { default?: unknown }).default;\n if (next === undefined) {\n return typeof candidate === \"object\"\n ? (candidate as EvalConfig)\n : undefined;\n }\n candidate = next;\n }\n return undefined;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // Each attempt builds a fresh driver, so a retry never inherits the failed\n // attempt's thread. (runWithRetries defines what counts as retryable.)\n return await runWithRetries(options.retries ?? 0, () =>\n runEval(def, {\n id,\n driver: createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n // Cap the driver turn at the eval's effective timeout (runEval's signal also aborts it).\n timeoutMs: def.timeoutMs ?? options.timeoutMs,\n }),\n strict: options.strict,\n row,\n timeoutMs: options.timeoutMs,\n }),\n );\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Run one already-loaded eval (from the `loaded` pre-pass), expanding a\n * dataset-driven eval into one run per row. Appends one result per row to\n * `results`, emitting `start`/`result` around each. Never throws: a load error\n * (carried in `loadError`) or a dataset-read failure surfaces as a non-passing\n * result. `total` counts eval files, not rows — per-row detail is carried in the\n * result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n def: EvalDefinition,\n loadError: string | undefined,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n // Load failed in the pre-pass (def is a placeholder) → one non-passing result.\n if (loadError) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: loadError,\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Base delay (ms) before the first retry; doubled per attempt, full-jittered, capped. */\nconst DEFAULT_RETRY_BASE_DELAY_MS = 250;\n/** Ceiling for a single retry backoff wait (ms). */\nconst MAX_RETRY_DELAY_MS = 5_000;\n\n/**\n * Run `attempt` up to `1 + retries` times, stopping as soon as it returns a\n * result that is neither a thrown error / per-eval timeout (`error`) nor a\n * transport/agent turn failure (`infraFailure`). Assertion failures set\n * neither, so a failed-but-completed eval is returned on the first try and\n * never retried. Returns the last result when every attempt failed on infra.\n *\n * Between attempts it waits a full-jittered exponential backoff (infra flakes\n * are overload-correlated). `retries` is coerced to a finite non-negative\n * integer; `baseDelayMs: 0` disables the wait (tests).\n */\nexport async function runWithRetries(\n retries: number,\n attempt: (attemptNumber: number) => Promise<EvalResult>,\n options: { baseDelayMs?: number } = {},\n): Promise<EvalResult> {\n const baseDelayMs = options.baseDelayMs ?? DEFAULT_RETRY_BASE_DELAY_MS;\n const maxRetries = Number.isFinite(retries)\n ? Math.max(0, Math.floor(retries))\n : 0;\n const maxAttempts = 1 + maxRetries;\n let result: EvalResult;\n for (let n = 1; ; n++) {\n result = await attempt(n);\n const infraFailed = result.error !== undefined || result.infraFailure;\n if (!infraFailed || n >= maxAttempts) return result;\n if (baseDelayMs > 0) {\n // Full jitter: a random wait in [0, min(cap, base * 2^(n-1))].\n const ceiling = Math.min(baseDelayMs * 2 ** (n - 1), MAX_RETRY_DELAY_MS);\n await sleep(Math.random() * ceiling);\n }\n }\n}\n\n/**\n * Whether an eval's `tags` satisfy a `--tag` filter: `true` when the filter is\n * empty/undefined (no filtering), otherwise only when the eval shares at least\n * one tag with it. An eval with no tags never matches a non-empty filter.\n */\nexport function matchesTags(\n defTags: string[] | undefined,\n filterTags: string[] | undefined,\n): boolean {\n if (!filterTags || filterTags.length === 0) return true;\n return defTags?.some((t) => filterTags.includes(t)) ?? false;\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Resolve the work-pool width: `--concurrency` wins; else the lowest\n * `maxConcurrency` any *participating* agent's `evals.config.ts` requests (all\n * evals share one per-user stream budget, so the most conservative ceiling\n * governs); else {@link DEFAULT_CONCURRENCY}.\n */\nexport function deriveConcurrency(\n activeAgents: Set<string>,\n configs: Map<string, EvalConfig>,\n cliConcurrency: number | undefined,\n): number {\n const configMin = [...configs.entries()]\n .filter(([agent]) => activeAgents.has(agent))\n .map(([, c]) => c.maxConcurrency)\n .filter((n): n is number => typeof n === \"number\")\n .reduce<number | undefined>(\n (min, n) => (min === undefined ? n : Math.min(min, n)),\n undefined,\n );\n return cliConcurrency ?? configMin ?? DEFAULT_CONCURRENCY;\n}\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n\n // Load each agent's `evals.config.ts` (best-effort, per-agent): its settings\n // apply only to that agent's evals. A malformed/missing config never aborts\n // the run — the agent just falls back to CLI options and built-in defaults.\n const configs = new Map<string, EvalConfig>();\n for (const c of discoverEvalConfigs(root)) {\n try {\n const cfg = await loadEvalConfig(c.file);\n if (cfg) configs.set(c.agent, cfg);\n } catch {\n // Ignore: fall back to CLI options / defaults for this agent.\n }\n }\n\n // Load each eval def and apply the `--tag` filter up front. Tags live on the\n // def, so a tag miss removes the eval entirely (like the substring filter\n // excludes files) rather than surfacing as a result. Load failures are kept\n // so a broken file still reports as a non-passing result.\n const loaded: Array<{\n d: DiscoveredEval;\n def: EvalDefinition;\n loadError?: string;\n }> = [];\n for (const d of discovered) {\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n loaded.push({\n d,\n // No def loaded; placeholder def is never run (error short-circuits).\n def: { test: () => {} },\n loadError: err instanceof Error ? err.message : String(err),\n });\n continue;\n }\n if (!matchesTags(def.tags, options.tags)) continue;\n loaded.push({ d, def });\n }\n\n // Pool width from participating agents' configs (see {@link deriveConcurrency}).\n const activeAgents = new Set(loaded.map((l) => l.d.agent));\n const concurrency = deriveConcurrency(\n activeAgents,\n configs,\n options.concurrency,\n );\n\n const total = loaded.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each loaded (tag-filtered) eval through the bounded pool — one in-flight\n // stream per eval, so the pool respects the server's per-user stream cap (see\n // mapPool/concurrency). A dataset eval expands into per-row runs that execute\n // serially within its slot; results preserve discovery order (mapPool writes\n // by index) and row order within each file. Per-agent timeout is folded into\n // the file's options (CLI wins over `evals.config.ts`; `def.timeoutMs` still\n // overrides, applied inside runEval). `total` counts eval files, not dataset\n // rows — per-row detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n loaded,\n concurrency,\n async ({ d, def, loadError }, index) => {\n const fileResults: EvalResult[] = [];\n const fileOptions: RunEvalsOptions = {\n ...options,\n timeoutMs: options.timeoutMs ?? configs.get(d.agent)?.timeoutMs,\n };\n await runDiscovered(\n d,\n def,\n loadError,\n index,\n total,\n runId,\n fileOptions,\n emit,\n fileResults,\n );\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAqGA,eAAe,aAAa,MAAgC;CAC1D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;AAEH,QAAO,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI;;;;;AAM5D,eAAe,SAAS,MAAuC;CAE7D,MAAM,MAAM,mBADA,MAAM,aAAa,KAAK,CACD;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;AAQT,eAAe,eAAe,MAA+C;AAE3E,QAAO,qBADK,MAAM,aAAa,KAAK,CACJ;;;;;;;AAQlC,SAAgB,qBAAqB,KAAsC;CACzE,IAAI,YAAqB;AAEzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;EACvC,MAAM,OAAQ,UAAoC;AAClD,MAAI,SAAS,OACX,QAAO,OAAO,cAAc,WACvB,YACD;AAEN,cAAY;;;;;;;;;AAWhB,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAGF,SAAO,MAAM,eAAe,QAAQ,WAAW,SAC7C,QAAQ,KAAK;GACX;GACA,QAAQ,iBAAiB;IACvB,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IAEb,WAAW,IAAI,aAAa,QAAQ;IACrC,CAAC;GACF,QAAQ,QAAQ;GAChB;GACA,WAAW,QAAQ;GACpB,CAAC,CACH;UACM,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;;AAYL,eAAe,cACb,GACA,KACA,WACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;AAG3B,KAAI,WAAW;AACb,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO;GACR;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,MAAM,8BAA8B;;AAEpC,MAAM,qBAAqB;;;;;;;;;;;;AAa3B,eAAsB,eACpB,SACA,SACA,UAAoC,EAAE,EACjB;CACrB,MAAM,cAAc,QAAQ,eAAe;CAI3C,MAAM,cAAc,KAHD,OAAO,SAAS,QAAQ,GACvC,KAAK,IAAI,GAAG,KAAK,MAAM,QAAQ,CAAC,GAChC;CAEJ,IAAI;AACJ,MAAK,IAAI,IAAI,IAAK,KAAK;AACrB,WAAS,MAAM,QAAQ,EAAE;AAEzB,MAAI,EADgB,OAAO,UAAU,UAAa,OAAO,iBACrC,KAAK,YAAa,QAAO;AAC7C,MAAI,cAAc,GAAG;GAEnB,MAAM,UAAU,KAAK,IAAI,cAAc,MAAM,IAAI,IAAI,mBAAmB;AACxE,SAAMA,WAAM,KAAK,QAAQ,GAAG,QAAQ;;;;;;;;;AAU1C,SAAgB,YACd,SACA,YACS;AACT,KAAI,CAAC,cAAc,WAAW,WAAW,EAAG,QAAO;AACnD,QAAO,SAAS,MAAM,MAAM,WAAW,SAAS,EAAE,CAAC,IAAI;;;AAIzD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;;AAQ5B,SAAgB,kBACd,cACA,SACA,gBACQ;CACR,MAAM,YAAY,CAAC,GAAG,QAAQ,SAAS,CAAC,CACrC,QAAQ,CAAC,WAAW,aAAa,IAAI,MAAM,CAAC,CAC5C,KAAK,GAAG,OAAO,EAAE,eAAe,CAChC,QAAQ,MAAmB,OAAO,MAAM,SAAS,CACjD,QACE,KAAK,MAAO,QAAQ,SAAY,IAAI,KAAK,IAAI,KAAK,EAAE,EACrD,OACD;AACH,QAAO,kBAAkB,aAAa;;;;;;;AAQxC,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CAKvC,MAAM,0BAAU,IAAI,KAAyB;AAC7C,MAAK,MAAM,KAAK,oBAAoB,KAAK,CACvC,KAAI;EACF,MAAM,MAAM,MAAM,eAAe,EAAE,KAAK;AACxC,MAAI,IAAK,SAAQ,IAAI,EAAE,OAAO,IAAI;SAC5B;CASV,MAAM,SAID,EAAE;AACP,MAAK,MAAM,KAAK,YAAY;EAC1B,IAAI;AACJ,MAAI;AACF,SAAM,MAAM,SAAS,EAAE,KAAK;WACrB,KAAK;AACZ,UAAO,KAAK;IACV;IAEA,KAAK,EAAE,YAAY,IAAI;IACvB,WAAW,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;IAC5D,CAAC;AACF;;AAEF,MAAI,CAAC,YAAY,IAAI,MAAM,QAAQ,KAAK,CAAE;AAC1C,SAAO,KAAK;GAAE;GAAG;GAAK,CAAC;;CAKzB,MAAM,cAAc,kBADC,IAAI,IAAI,OAAO,KAAK,MAAM,EAAE,EAAE,MAAM,CAAC,EAGxD,SACA,QAAQ,YACT;CAED,MAAM,QAAQ,OAAO;AACrB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkCtC,MAAM,WAvBU,MAAM,QACpB,QACA,aACA,OAAO,EAAE,GAAG,KAAK,aAAa,UAAU;GACtC,MAAM,cAA4B,EAAE;GACpC,MAAM,cAA+B;IACnC,GAAG;IACH,WAAW,QAAQ,aAAa,QAAQ,IAAI,EAAE,MAAM,EAAE;IACvD;AACD,SAAM,cACJ,GACA,KACA,WACA,OACA,OACA,OACA,aACA,MACA,YACD;AACD,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
@@ -51,6 +51,11 @@ interface DriveResult {
51
51
  reply: string;
52
52
  /** Names of tools the agent called during the turn. */
53
53
  toolCalls: string[];
54
+ /** Tool calls with their parsed arguments, in call order. */
55
+ toolCallDetails: Array<{
56
+ name: string;
57
+ args: Record<string, unknown>;
58
+ }>;
54
59
  /** Whether the turn completed without an agent/stream error. */
55
60
  succeeded: boolean;
56
61
  /** Thread/session id, when the driver exposes one. */
@@ -63,7 +68,14 @@ interface DriveResult {
63
68
  * app's agents endpoint; future drivers (in-process) implement the same shape.
64
69
  */
65
70
  interface EvalDriver {
66
- send(message: string): Promise<DriveResult>;
71
+ /**
72
+ * Drive one turn. `options.signal`, when provided, aborts the in-flight turn:
73
+ * the runner passes its per-eval timeout signal so a timed-out eval cancels
74
+ * the request instead of leaking a live stream.
75
+ */
76
+ send(message: string, options?: {
77
+ signal?: AbortSignal;
78
+ }): Promise<DriveResult>;
67
79
  /**
68
80
  * Drop the current conversation so the next `send` starts a fresh thread.
69
81
  * Optional: drivers without a session concept omit it.
@@ -100,6 +112,13 @@ interface TestContext {
100
112
  succeeded(): AssertionHandle;
101
113
  /** Assert a tool was called during the run (gate by default). */
102
114
  calledTool(name: string): AssertionHandle;
115
+ /**
116
+ * Assert a tool was called with arguments that deep-contain `expected`: every
117
+ * key in `expected` must equal the actual argument (recursively for nested
118
+ * objects; arrays match element-for-element), so extra arguments are ignored.
119
+ * Gate by default.
120
+ */
121
+ calledToolWith(name: string, expected: Record<string, unknown>): AssertionHandle;
103
122
  /** Assert a value against a matcher, e.g. `t.check(t.reply, includes("Sunny"))`. */
104
123
  check(value: string, matcher: Matcher): AssertionHandle;
105
124
  /**
@@ -129,6 +148,13 @@ interface EvalDefinition {
129
148
  description?: string;
130
149
  /** Target agent id. Defaults to the eval's parent `server/agents/<id>` dir. */
131
150
  agent?: string;
151
+ /** Free-form tags for filtering (see the runner's `tags` / `--tag` option). */
152
+ tags?: string[];
153
+ /**
154
+ * Per-eval timeout (ms): `runEval` races the test against it and records a
155
+ * non-passing result instead of hanging. Overrides the runner/CLI default.
156
+ */
157
+ timeoutMs?: number;
132
158
  /**
133
159
  * Run this eval once per row of a Databricks managed evaluation dataset (a
134
160
  * Unity Catalog `catalog.schema.table` with `inputs`/`expectations` columns).
@@ -142,6 +168,13 @@ interface EvalDefinition {
142
168
  /** The eval body: drive the agent and assert on its behavior. */
143
169
  test(t: TestContext): Promise<void> | void;
144
170
  }
171
+ /** Per-directory config from `evals.config.ts` (see {@link defineEvalConfig}). */
172
+ interface EvalConfig {
173
+ /** Max evals to run concurrently. */
174
+ maxConcurrency?: number;
175
+ /** Default per-eval timeout. */
176
+ timeoutMs?: number;
177
+ }
145
178
  /** The outcome of running one eval. */
146
179
  interface EvalResult {
147
180
  id: string;
@@ -155,9 +188,14 @@ interface EvalResult {
155
188
  passed: boolean;
156
189
  /** Set when the eval threw before completing. */
157
190
  error?: string;
191
+ /**
192
+ * A turn failed at the transport/agent level (`succeeded: false`), not on an
193
+ * assertion — a retryable infra flake, distinct from `error`.
194
+ */
195
+ infraFailure?: boolean;
158
196
  /** MLflow trace id of the eval's last turn, for attaching assessments. */
159
197
  traceId?: string;
160
198
  }
161
199
  //#endregion
162
- export { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext };
200
+ export { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalConfig, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext };
163
201
  //# sourceMappingURL=types.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;EAQoB;EAN5B,IAAA,IAAQ,eAAA;EAMmC;;;;;EAA3C,OAAA,CAAQ,SAAA,WAAoB,eAAA;AAAA;;UAIb,WAAA;EAJ4B;EAM3C,KAAA;EAF0B;EAI1B,SAAA;EAJ0B;EAM1B,SAAA;EAFA;EAIA,SAAA;EAAA;EAEA,OAAA;AAAA;;AAOF;;;UAAiB,UAAA;EACf,IAAA,CAAK,OAAA,WAAkB,OAAA,CAAQ,WAAA;EAA1B;;;;EAKL,KAAA;AAAA;AAIF;AAAA,UAAiB,WAAA;;EAEf,IAAA,CAAK,OAAA,WAAkB,OAAA;EAiBP;;;;;EAXhB,KAAA;EAgCwC;EAAA,SA9B/B,KAAA;EAgC6B;EAAA,SA9B7B,SAAA;EAgCM;EAAA,SA9BN,SAAA;EA8BwB;;;;EAAA,SAzBxB,KAAA,EAAO,MAAA;EAjBO;;;;EAAA,SAsBd,QAAA,EAAU,MAAA;EALV;EAOT,SAAA,IAAa,eAAA;EAFJ;EAIT,UAAA,CAAW,IAAA,WAAe,eAAA;EAF1B;EAIA,KAAA,CAAM,KAAA,UAAe,OAAA,EAAS,OAAA,GAAU,eAAA;EAFxC;;;;;;;EAUA,KAAA;IAAA,mEAEE,UAAA,CAAW,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAA3B;IAEX,QAAA,CAAS,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAFE;IAItC,MAAA,CAAO,IAAA,EAAM,eAAA,GAAkB,OAAA,CAAQ,eAAA;EAAA;EAFX;EAK9B,IAAA,CAAK,MAAA;AAAA;;UAIU,eAAA;EACf,IAAA;EACA,cAAA;EACA,YAAA,EAAc,MAAA;AAAA;;UAIC,cAAA;EAPA;EASf,WAAA;;EAEA,KAAA;EAVA;;;;;;EAiBA,OAAA;IAAY,KAAA;IAAe,KAAA;EAAA;EAT3B;EAWA,IAAA,CAAK,CAAA,EAAG,WAAA,GAAc,OAAA;AAAA;;UAIP,UAAA;EACf,EAAA;EACA,WAAA;EANK;EAQL,OAAA;IAAY,MAAA;EAAA;EACZ,UAAA,EAAY,eAAA;EALa;EAOzB,MAAA;EAF2B;EAI3B,KAAA;EAPA;EASA,OAAA;AAAA"}
1
+ {"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;EAQoB;EAN5B,IAAA,IAAQ,eAAA;EAMmC;;;;;EAA3C,OAAA,CAAQ,SAAA,WAAoB,eAAA;AAAA;;UAIb,WAAA;EAJ4B;EAM3C,KAAA;EAF0B;EAI1B,SAAA;EAEsB;EAAtB,eAAA,EAAiB,KAAA;IAAQ,IAAA;IAAc,IAAA,EAAM,MAAA;EAAA;EAApB;EAEzB,SAAA;EAF6C;EAI7C,SAAA;EAAA;EAEA,OAAA;AAAA;;AAOF;;;UAAiB,UAAA;EASJ;;;;;EAHX,IAAA,CACE,OAAA,UACA,OAAA;IAAY,MAAA,GAAS,WAAA;EAAA,IACpB,OAAA,CAAQ,WAAA;EADT;;;;EAMF,KAAA;AAAA;AAIF;AAAA,UAAiB,WAAA;;EAEf,IAAA,CAAK,OAAA,WAAkB,OAAA;EAiBP;;;;;EAXhB,KAAA;EAgC8B;EAAA,SA9BrB,KAAA;EAwC+B;EAAA,SAtC/B,SAAA;EAwC6B;EAAA,SAtC7B,SAAA;EAwCM;;;;EAAA,SAnCN,KAAA,EAAO,MAAA;EAjBhB;;;;EAAA,SAsBS,QAAA,EAAU,MAAA;EAZV;EAcT,SAAA,IAAa,eAAA;EAPJ;EAST,UAAA,CAAW,IAAA,WAAe,eAAA;EAJjB;;;;;;EAWT,cAAA,CACE,IAAA,UACA,QAAA,EAAU,MAAA,oBACT,eAAA;EAHH;EAKA,KAAA,CAAM,KAAA,UAAe,OAAA,EAAS,OAAA,GAAU,eAAA;EAH5B;;;;;;;EAWZ,KAAA;IAAA,mEAEE,UAAA,CAAW,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAA3B;IAEX,QAAA,CAAS,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAFE;IAItC,MAAA,CAAO,IAAA,EAAM,eAAA,GAAkB,OAAA,CAAQ,eAAA;EAAA;EAFX;EAK9B,IAAA,CAAK,MAAA;AAAA;;UAIU,eAAA;EACf,IAAA;EACA,cAAA;EACA,YAAA,EAAc,MAAA;AAAA;;UAIC,cAAA;EAPA;EASf,WAAA;;EAEA,KAAA;EAVA;EAYA,IAAA;EAVA;;;;EAeA,SAAA;EAX6B;;;;;;EAkB7B,OAAA;IAAY,KAAA;IAAe,KAAA;EAAA;EAE3B;EAAA,IAAA,CAAK,CAAA,EAAG,WAAA,GAAc,OAAA;AAAA;;UAIP,UAAA;EAJc;EAM7B,cAAA;EAFyB;EAIzB,SAAA;AAAA;;UAIe,UAAA;EACf,EAAA;EACA,WAAA;EAG2B;EAD3B,OAAA;IAAY,MAAA;EAAA;EACZ,UAAA,EAAY,eAAA;EAAZ;EAEA,MAAA;EAAA;EAEA,KAAA;EAKA;;;;EAAA,YAAA;;EAEA,OAAA;AAAA"}
@@ -53,8 +53,8 @@ declare function getPluginManifest(plugin: PluginConstructor): PluginManifest;
53
53
  */
54
54
  declare function getResourceRequirements(plugin: PluginConstructor): {
55
55
  required: boolean;
56
- description: string;
57
56
  type: ResourceType;
57
+ description: string;
58
58
  permission: ResourcePermission;
59
59
  alias: string;
60
60
  resourceKey: string;
@@ -0,0 +1,18 @@
1
+ # Function: defineEvalConfig()
2
+
3
+ ```ts
4
+ function defineEvalConfig(config: EvalConfig): EvalConfig;
5
+
6
+ ```
7
+
8
+ Define per-directory eval config. Default-export from `evals.config.ts`.
9
+
10
+ ## Parameters[​](#parameters "Direct link to Parameters")
11
+
12
+ | Parameter | Type |
13
+ | --------- | ------------ |
14
+ | `config` | `EvalConfig` |
15
+
16
+ ## Returns[​](#returns "Direct link to Returns")
17
+
18
+ `EvalConfig`
@@ -0,0 +1,18 @@
1
+ # Function: discoverEvalConfigs()
2
+
3
+ ```ts
4
+ function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[];
5
+
6
+ ```
7
+
8
+ Discover the per-agent `evals.config.ts` (from [defineEvalConfig](./docs/api/appkit/Function.defineEvalConfig.md)) at `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent: each agent's config applies only to that agent's evals. Agents without a config file are omitted. Returns a stable, sorted list.
9
+
10
+ ## Parameters[​](#parameters "Direct link to Parameters")
11
+
12
+ | Parameter | Type |
13
+ | --------- | -------- |
14
+ | `rootDir` | `string` |
15
+
16
+ ## Returns[​](#returns "Direct link to Returns")
17
+
18
+ [`DiscoveredEvalConfig`](./docs/api/appkit/Interface.DiscoveredEvalConfig.md)\[]
@@ -0,0 +1,18 @@
1
+ # Function: formatResultsJUnit()
2
+
3
+ ```ts
4
+ function formatResultsJUnit(results: EvalResult[]): string;
5
+
6
+ ```
7
+
8
+ Render results as JUnit XML for standard CI test reporters: a single `<testsuite name="appkit-agent-evals">` with one `<testcase>` per result. Failures carry a `<failure>` (error or failing-gate summary); skips a `<skipped>`. All attribute/text values are XML-escaped.
9
+
10
+ ## Parameters[​](#parameters "Direct link to Parameters")
11
+
12
+ | Parameter | Type |
13
+ | --------- | ------------------------------------------------------------------ |
14
+ | `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
15
+
16
+ ## Returns[​](#returns "Direct link to Returns")
17
+
18
+ `string`
@@ -0,0 +1,18 @@
1
+ # Function: formatResultsJson()
2
+
3
+ ```ts
4
+ function formatResultsJson(results: EvalResult[]): string;
5
+
6
+ ```
7
+
8
+ Render results as a machine-readable JSON report (2-space indented): `{ summary: EvalSummary, results: EvalResult[] }`. Faithful to the types — every field present on a result round-trips.
9
+
10
+ ## Parameters[​](#parameters "Direct link to Parameters")
11
+
12
+ | Parameter | Type |
13
+ | --------- | ------------------------------------------------------------------ |
14
+ | `results` | [`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)\[] |
15
+
16
+ ## Returns[​](#returns "Direct link to Returns")
17
+
18
+ `string`
@@ -0,0 +1,28 @@
1
+ # Function: runWithRetries()
2
+
3
+ ```ts
4
+ function runWithRetries(
5
+ retries: number,
6
+ attempt: (attemptNumber: number) => Promise<EvalResult>,
7
+ options: {
8
+ baseDelayMs?: number;
9
+ }): Promise<EvalResult>;
10
+
11
+ ```
12
+
13
+ Run `attempt` up to `1 + retries` times, stopping as soon as it returns a result that is neither a thrown error / per-eval timeout (`error`) nor a transport/agent turn failure (`infraFailure`). Assertion failures set neither, so a failed-but-completed eval is returned on the first try and never retried. Returns the last result when every attempt failed on infra.
14
+
15
+ Between attempts it waits a full-jittered exponential backoff (infra flakes are overload-correlated). `retries` is coerced to a finite non-negative integer; `baseDelayMs: 0` disables the wait (tests).
16
+
17
+ ## Parameters[​](#parameters "Direct link to Parameters")
18
+
19
+ | Parameter | Type |
20
+ | ---------------------- | --------------------------------------------------------------------------------------------------------- |
21
+ | `retries` | `number` |
22
+ | `attempt` | (`attemptNumber`: `number`) => `Promise`<[`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)> |
23
+ | `options` | { `baseDelayMs?`: `number`; } |
24
+ | `options.baseDelayMs?` | `number` |
25
+
26
+ ## Returns[​](#returns "Direct link to Returns")
27
+
28
+ `Promise`<[`EvalResult`](./docs/api/appkit/Interface.EvalResult.md)>
@@ -0,0 +1,20 @@
1
+ # Function: userTurns()
2
+
3
+ ```ts
4
+ function userTurns(input: Record<string, unknown>): string[];
5
+
6
+ ```
7
+
8
+ Extract every user-message content, in order, from an MLflow `{"messages":[{"role":"user","content":"..."}]}` input. A dataset row can carry a full multi-turn conversation; replaying these against one thread (one `t.send` per returned string) lets the agent see the accumulating history.
9
+
10
+ Only `role === "user"` turns are returned — any interleaved `assistant`/ `system` messages in the row are ignored, since the agent generates its own responses; you never inject the dataset's assistant turns. A single-user-turn row yields a one-element array (backward compatible); a row with no `messages` yields `[]`.
11
+
12
+ ## Parameters[​](#parameters "Direct link to Parameters")
13
+
14
+ | Parameter | Type |
15
+ | --------- | ----------------------------- |
16
+ | `input` | `Record`<`string`, `unknown`> |
17
+
18
+ ## Returns[​](#returns "Direct link to Returns")
19
+
20
+ `string`\[]
@@ -0,0 +1,25 @@
1
+ # Interface: DiscoveredEvalConfig
2
+
3
+ A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`.
4
+
5
+ ## Properties[​](#properties "Direct link to Properties")
6
+
7
+ ### agent[​](#agent "Direct link to agent")
8
+
9
+ ```ts
10
+ agent: string;
11
+
12
+ ```
13
+
14
+ The agent id whose evals this config applies to.
15
+
16
+ ***
17
+
18
+ ### file[​](#file "Direct link to file")
19
+
20
+ ```ts
21
+ file: string;
22
+
23
+ ```
24
+
25
+ Absolute path to the `evals.config.ts` file.
@@ -37,6 +37,34 @@ Whether the turn completed without an agent/stream error.
37
37
 
38
38
  ***
39
39
 
40
+ ### toolCallDetails[​](#toolcalldetails "Direct link to toolCallDetails")
41
+
42
+ ```ts
43
+ toolCallDetails: {
44
+ args: Record<string, unknown>;
45
+ name: string;
46
+ }[];
47
+
48
+ ```
49
+
50
+ Tool calls with their parsed arguments, in call order.
51
+
52
+ #### args[​](#args "Direct link to args")
53
+
54
+ ```ts
55
+ args: Record<string, unknown>;
56
+
57
+ ```
58
+
59
+ #### name[​](#name "Direct link to name")
60
+
61
+ ```ts
62
+ name: string;
63
+
64
+ ```
65
+
66
+ ***
67
+
40
68
  ### toolCalls[​](#toolcalls "Direct link to toolCalls")
41
69
 
42
70
  ```ts