@databricks/appkit 0.74.1 → 0.75.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CLAUDE.md +4 -1
  2. package/dist/appkit/package.js +1 -1
  3. package/dist/beta.d.ts +5 -5
  4. package/dist/beta.js +5 -4
  5. package/dist/cache/index.js +2 -2
  6. package/dist/cache/storage/persistent.js +1 -1
  7. package/dist/cli/commands/agent/eval.js +78 -6
  8. package/dist/cli/commands/agent/eval.js.map +1 -1
  9. package/dist/cli/commands/registry/add.js +1 -1
  10. package/dist/cli/commands/registry/config-writer.js +1 -1
  11. package/dist/cli/index.js +4 -1
  12. package/dist/cli/index.js.map +1 -1
  13. package/dist/connectors/index.js +1 -1
  14. package/dist/connectors/jobs/client.js +1 -1
  15. package/dist/connectors/lakebase/index.d.ts +2 -2
  16. package/dist/connectors/lakebase/index.d.ts.map +1 -1
  17. package/dist/connectors/lakebase/index.js +25 -2
  18. package/dist/connectors/lakebase/index.js.map +1 -1
  19. package/dist/connectors/mcp/client.js +1 -1
  20. package/dist/connectors/sql-warehouse/client.js +1 -1
  21. package/dist/context/service-context.js +1 -1
  22. package/dist/core/appkit.js +1 -1
  23. package/dist/database/errors.js +1 -1
  24. package/dist/evals/discover.d.ts +8 -1
  25. package/dist/evals/discover.d.ts.map +1 -1
  26. package/dist/evals/discover.js +11 -1
  27. package/dist/evals/discover.js.map +1 -1
  28. package/dist/evals/index.d.ts +3 -3
  29. package/dist/evals/index.js +2 -2
  30. package/dist/evals/run-evals.d.ts +9 -2
  31. package/dist/evals/run-evals.d.ts.map +1 -1
  32. package/dist/evals/run-evals.js +13 -2
  33. package/dist/evals/run-evals.js.map +1 -1
  34. package/dist/evals/types.d.ts +40 -2
  35. package/dist/evals/types.d.ts.map +1 -1
  36. package/dist/index.js +1 -1
  37. package/dist/plugin/dev-reader.js +1 -1
  38. package/dist/plugin/interceptors/retry.js +1 -1
  39. package/dist/plugin/plugin.js +1 -1
  40. package/dist/plugins/analytics/analytics.js +1 -1
  41. package/dist/plugins/analytics/result-delivery.js +1 -1
  42. package/dist/plugins/beta-exports.generated.d.ts +1 -3
  43. package/dist/plugins/beta-exports.generated.js +0 -2
  44. package/dist/plugins/database/config.js +14 -0
  45. package/dist/plugins/database/config.js.map +1 -0
  46. package/dist/plugins/database/database.d.ts +22 -5
  47. package/dist/plugins/database/database.d.ts.map +1 -1
  48. package/dist/plugins/database/database.js +23 -8
  49. package/dist/plugins/database/database.js.map +1 -1
  50. package/dist/plugins/database/lifecycle.js +3 -3
  51. package/dist/plugins/database/lifecycle.js.map +1 -1
  52. package/dist/plugins/database/load-schema.js +40 -0
  53. package/dist/plugins/database/load-schema.js.map +1 -0
  54. package/dist/plugins/database/manifest.js +1 -1
  55. package/dist/plugins/database/types.d.ts +12 -3
  56. package/dist/plugins/database/types.d.ts.map +1 -1
  57. package/dist/plugins/files/plugin.js +1 -1
  58. package/dist/plugins/jobs/plugin.js +1 -1
  59. package/dist/plugins/lakebase/lakebase.js +1 -1
  60. package/dist/plugins/server/index.js +1 -1
  61. package/dist/plugins/server/vite-dev-server.js +1 -1
  62. package/dist/registry/manifest-loader.d.ts +1 -1
  63. package/dist/registry/manifest-loader.js +1 -1
  64. package/dist/registry/resource-registry.js +1 -1
  65. package/dist/shared/src/schemas/manifest.d.ts +33 -33
  66. package/dist/stream/arrow-stream-processor.js +1 -1
  67. package/dist/stream/stream-manager.js +1 -1
  68. package/docs/api/appkit/Function.database.md +30 -29
  69. package/docs/api/appkit/Function.findRootEvalConfig.md +18 -0
  70. package/docs/api/appkit/Function.loadRootEvalConfig.md +18 -0
  71. package/docs/api/appkit/Interface.EvalWebServer.md +47 -0
  72. package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +8 -6
  73. package/docs/api/appkit.md +4 -1
  74. package/docs/plugins/database.md +46 -18
  75. package/llms.txt +4 -1
  76. package/package.json +1 -1
  77. package/sbom.cdx.json +1 -1
@@ -1,6 +1,6 @@
1
- import { createLogger } from "../logging/logger.js";
2
1
  import { AppKitError } from "../errors/base.js";
3
2
  import "../errors/index.js";
3
+ import { createLogger } from "../logging/logger.js";
4
4
 
5
5
  //#region src/database/errors.ts
6
6
  const logger = createLogger("database");
@@ -22,6 +22,13 @@ interface DiscoveredEvalConfig {
22
22
  * path relative to that evals dir with `.eval.ts` stripped. Sorted + stable.
23
23
  */
24
24
  declare function discoverEvalFiles(rootDir: string): DiscoveredEval[];
25
+ /**
26
+ * Path to the root `evals.config.ts` (from {@link defineEvalConfig}) at
27
+ * `<rootDir>/evals.config.ts`, or `undefined` when absent. The root config
28
+ * holds run-wide settings (`baseUrl`, `webServer`); it's distinct from the
29
+ * per-agent configs found by {@link discoverEvalConfigs}.
30
+ */
31
+ declare function findRootEvalConfig(rootDir: string): string | undefined;
25
32
  /**
26
33
  * Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at
27
34
  * `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:
@@ -30,5 +37,5 @@ declare function discoverEvalFiles(rootDir: string): DiscoveredEval[];
30
37
  */
31
38
  declare function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[];
32
39
  //#endregion
33
- export { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles };
40
+ export { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig };
34
41
  //# sourceMappingURL=discover.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"discover.d.ts","names":[],"sources":["../../src/evals/discover.ts"],"mappings":";;UAOiB,cAAA;EAAc;EAE7B,IAAA;EAF6B;EAI7B,EAAA;EAAA;EAEA,KAAA;AAAA;;UAIe,oBAAA;EAAoB;EAEnC,IAAA;EAAA;EAEA,KAAA;AAAA;;;;;AA+DF;;iBA3BgB,iBAAA,CAAkB,OAAA,WAAkB,cAAA;;;;;;;iBA2BpC,mBAAA,CAAoB,OAAA,WAAkB,oBAAA"}
1
+ {"version":3,"file":"discover.d.ts","names":[],"sources":["../../src/evals/discover.ts"],"mappings":";;UAOiB,cAAA;EAAc;EAE7B,IAAA;EAF6B;EAI7B,EAAA;EAAA;EAEA,KAAA;AAAA;;UAIe,oBAAA;EAAoB;EAEnC,IAAA;EAAA;EAEA,KAAA;AAAA;;;;;AA+DF;;iBA3BgB,iBAAA,CAAkB,OAAA,WAAkB,cAAA;;;AAsCpD;;;;iBAXgB,kBAAA,CAAmB,OAAA;;;;;;;iBAWnB,mBAAA,CAAoB,OAAA,WAAkB,oBAAA"}
@@ -56,6 +56,16 @@ function discoverEvalFiles(rootDir) {
56
56
  return out.sort((a, b) => a.agent.localeCompare(b.agent) || a.id.localeCompare(b.id));
57
57
  }
58
58
  /**
59
+ * Path to the root `evals.config.ts` (from {@link defineEvalConfig}) at
60
+ * `<rootDir>/evals.config.ts`, or `undefined` when absent. The root config
61
+ * holds run-wide settings (`baseUrl`, `webServer`); it's distinct from the
62
+ * per-agent configs found by {@link discoverEvalConfigs}.
63
+ */
64
+ function findRootEvalConfig(rootDir) {
65
+ const file = path.join(rootDir, "evals.config.ts");
66
+ return existsSync(file) ? file : void 0;
67
+ }
68
+ /**
59
69
  * Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at
60
70
  * `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:
61
71
  * each agent's config applies only to that agent's evals. Agents without a
@@ -75,5 +85,5 @@ function discoverEvalConfigs(rootDir) {
75
85
  }
76
86
 
77
87
  //#endregion
78
- export { discoverEvalConfigs, discoverEvalFiles };
88
+ export { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig };
79
89
  //# sourceMappingURL=discover.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"discover.js","names":[],"sources":["../../src/evals/discover.ts"],"sourcesContent":["import { existsSync, readdirSync } from \"node:fs\";\nimport path from \"node:path\";\n\nimport { agentDirNames } from \"../core/agent/agent-dirs\";\nimport { CODE_AGENTS_SOURCE_DIR } from \"../core/agent/load-code-agents\";\n\n/** An eval file found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEval {\n /** Absolute path to the `*.eval.ts` file. */\n file: string;\n /** Id relative to the agent's evals dir, without `.eval.ts` (e.g. `weather/basic`). */\n id: string;\n /** The agent id (the `server/agents/<agent>` directory name). */\n agent: string;\n}\n\n/** A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEvalConfig {\n /** Absolute path to the `evals.config.ts` file. */\n file: string;\n /** The agent id whose evals this config applies to. */\n agent: string;\n}\n\n/** Recursively collect `*.eval.ts` files under `dir`. Empty when `dir` is absent. */\nfunction evalFilesIn(dir: string): string[] {\n try {\n return readdirSync(dir, { recursive: true, withFileTypes: true })\n .filter((e) => e.isFile() && e.name.endsWith(\".eval.ts\"))\n .map((e) => path.join(e.parentPath, e.name));\n } catch {\n return [];\n }\n}\n\n/**\n * List the agent directory names under `<rootDir>/server/agents/` (empty if the\n * dir is absent). Shared by the eval-file and eval-config discovery below.\n */\nfunction listAgents(rootDir: string): { agentsDir: string; agents: string[] } {\n const agentsDir = path.join(rootDir, CODE_AGENTS_SOURCE_DIR);\n try {\n return {\n agentsDir,\n agents: agentDirNames(readdirSync(agentsDir, { withFileTypes: true })),\n };\n } catch {\n return { agentsDir, agents: [] };\n }\n}\n\n/**\n * Discover evals under `<rootDir>/server/agents/<agent>/evals/` — co-located\n * with each agent's `agent.{md,ts}` (same folder-per-agent layout the agents\n * plugin discovers). The agent id is the folder name; the eval id is the file\n * path relative to that evals dir with `.eval.ts` stripped. Sorted + stable.\n */\nexport function discoverEvalFiles(rootDir: string): DiscoveredEval[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEval[] = [];\n\n for (const agent of agents) {\n const evalsDir = path.join(agentsDir, agent, \"evals\");\n for (const file of evalFilesIn(evalsDir)) {\n const id = path\n .relative(evalsDir, file)\n .replace(/\\.eval\\.ts$/, \"\")\n .split(path.sep)\n .join(\"/\");\n out.push({ file, id, agent });\n }\n }\n\n return out.sort(\n (a, b) => a.agent.localeCompare(b.agent) || a.id.localeCompare(b.id),\n );\n}\n\n/**\n * Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at\n * `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:\n * each agent's config applies only to that agent's evals. Agents without a\n * config file are omitted. Returns a stable, sorted list.\n */\nexport function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEvalConfig[] = [];\n\n for (const agent of agents) {\n const file = path.join(agentsDir, agent, \"evals\", \"evals.config.ts\");\n if (existsSync(file)) out.push({ file, agent });\n }\n\n return out.sort((a, b) => a.agent.localeCompare(b.agent));\n}\n"],"mappings":";;;;;;;AAyBA,SAAS,YAAY,KAAuB;AAC1C,KAAI;AACF,SAAO,YAAY,KAAK;GAAE,WAAW;GAAM,eAAe;GAAM,CAAC,CAC9D,QAAQ,MAAM,EAAE,QAAQ,IAAI,EAAE,KAAK,SAAS,WAAW,CAAC,CACxD,KAAK,MAAM,KAAK,KAAK,EAAE,YAAY,EAAE,KAAK,CAAC;SACxC;AACN,SAAO,EAAE;;;;;;;AAQb,SAAS,WAAW,SAA0D;CAC5E,MAAM,YAAY,KAAK,KAAK,SAAS,uBAAuB;AAC5D,KAAI;AACF,SAAO;GACL;GACA,QAAQ,cAAc,YAAY,WAAW,EAAE,eAAe,MAAM,CAAC,CAAC;GACvE;SACK;AACN,SAAO;GAAE;GAAW,QAAQ,EAAE;GAAE;;;;;;;;;AAUpC,SAAgB,kBAAkB,SAAmC;CACnE,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAAwB,EAAE;AAEhC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,WAAW,KAAK,KAAK,WAAW,OAAO,QAAQ;AACrD,OAAK,MAAM,QAAQ,YAAY,SAAS,EAAE;GACxC,MAAM,KAAK,KACR,SAAS,UAAU,KAAK,CACxB,QAAQ,eAAe,GAAG,CAC1B,MAAM,KAAK,IAAI,CACf,KAAK,IAAI;AACZ,OAAI,KAAK;IAAE;IAAM;IAAI;IAAO,CAAC;;;AAIjC,QAAO,IAAI,MACR,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,IAAI,EAAE,GAAG,cAAc,EAAE,GAAG,CACrE;;;;;;;;AASH,SAAgB,oBAAoB,SAAyC;CAC3E,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAA8B,EAAE;AAEtC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,OAAO,KAAK,KAAK,WAAW,OAAO,SAAS,kBAAkB;AACpE,MAAI,WAAW,KAAK,CAAE,KAAI,KAAK;GAAE;GAAM;GAAO,CAAC;;AAGjD,QAAO,IAAI,MAAM,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,CAAC"}
1
+ {"version":3,"file":"discover.js","names":[],"sources":["../../src/evals/discover.ts"],"sourcesContent":["import { existsSync, readdirSync } from \"node:fs\";\nimport path from \"node:path\";\n\nimport { agentDirNames } from \"../core/agent/agent-dirs\";\nimport { CODE_AGENTS_SOURCE_DIR } from \"../core/agent/load-code-agents\";\n\n/** An eval file found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEval {\n /** Absolute path to the `*.eval.ts` file. */\n file: string;\n /** Id relative to the agent's evals dir, without `.eval.ts` (e.g. `weather/basic`). */\n id: string;\n /** The agent id (the `server/agents/<agent>` directory name). */\n agent: string;\n}\n\n/** A per-agent `evals.config.ts` found under `server/agents/<agent>/evals/`. */\nexport interface DiscoveredEvalConfig {\n /** Absolute path to the `evals.config.ts` file. */\n file: string;\n /** The agent id whose evals this config applies to. */\n agent: string;\n}\n\n/** Recursively collect `*.eval.ts` files under `dir`. Empty when `dir` is absent. */\nfunction evalFilesIn(dir: string): string[] {\n try {\n return readdirSync(dir, { recursive: true, withFileTypes: true })\n .filter((e) => e.isFile() && e.name.endsWith(\".eval.ts\"))\n .map((e) => path.join(e.parentPath, e.name));\n } catch {\n return [];\n }\n}\n\n/**\n * List the agent directory names under `<rootDir>/server/agents/` (empty if the\n * dir is absent). Shared by the eval-file and eval-config discovery below.\n */\nfunction listAgents(rootDir: string): { agentsDir: string; agents: string[] } {\n const agentsDir = path.join(rootDir, CODE_AGENTS_SOURCE_DIR);\n try {\n return {\n agentsDir,\n agents: agentDirNames(readdirSync(agentsDir, { withFileTypes: true })),\n };\n } catch {\n return { agentsDir, agents: [] };\n }\n}\n\n/**\n * Discover evals under `<rootDir>/server/agents/<agent>/evals/` — co-located\n * with each agent's `agent.{md,ts}` (same folder-per-agent layout the agents\n * plugin discovers). The agent id is the folder name; the eval id is the file\n * path relative to that evals dir with `.eval.ts` stripped. Sorted + stable.\n */\nexport function discoverEvalFiles(rootDir: string): DiscoveredEval[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEval[] = [];\n\n for (const agent of agents) {\n const evalsDir = path.join(agentsDir, agent, \"evals\");\n for (const file of evalFilesIn(evalsDir)) {\n const id = path\n .relative(evalsDir, file)\n .replace(/\\.eval\\.ts$/, \"\")\n .split(path.sep)\n .join(\"/\");\n out.push({ file, id, agent });\n }\n }\n\n return out.sort(\n (a, b) => a.agent.localeCompare(b.agent) || a.id.localeCompare(b.id),\n );\n}\n\n/**\n * Path to the root `evals.config.ts` (from {@link defineEvalConfig}) at\n * `<rootDir>/evals.config.ts`, or `undefined` when absent. The root config\n * holds run-wide settings (`baseUrl`, `webServer`); it's distinct from the\n * per-agent configs found by {@link discoverEvalConfigs}.\n */\nexport function findRootEvalConfig(rootDir: string): string | undefined {\n const file = path.join(rootDir, \"evals.config.ts\");\n return existsSync(file) ? file : undefined;\n}\n\n/**\n * Discover the per-agent `evals.config.ts` (from {@link defineEvalConfig}) at\n * `<rootDir>/server/agents/<agent>/evals/evals.config.ts`. Config is per-agent:\n * each agent's config applies only to that agent's evals. Agents without a\n * config file are omitted. Returns a stable, sorted list.\n */\nexport function discoverEvalConfigs(rootDir: string): DiscoveredEvalConfig[] {\n const { agentsDir, agents } = listAgents(rootDir);\n const out: DiscoveredEvalConfig[] = [];\n\n for (const agent of agents) {\n const file = path.join(agentsDir, agent, \"evals\", \"evals.config.ts\");\n if (existsSync(file)) out.push({ file, agent });\n }\n\n return out.sort((a, b) => a.agent.localeCompare(b.agent));\n}\n"],"mappings":";;;;;;;AAyBA,SAAS,YAAY,KAAuB;AAC1C,KAAI;AACF,SAAO,YAAY,KAAK;GAAE,WAAW;GAAM,eAAe;GAAM,CAAC,CAC9D,QAAQ,MAAM,EAAE,QAAQ,IAAI,EAAE,KAAK,SAAS,WAAW,CAAC,CACxD,KAAK,MAAM,KAAK,KAAK,EAAE,YAAY,EAAE,KAAK,CAAC;SACxC;AACN,SAAO,EAAE;;;;;;;AAQb,SAAS,WAAW,SAA0D;CAC5E,MAAM,YAAY,KAAK,KAAK,SAAS,uBAAuB;AAC5D,KAAI;AACF,SAAO;GACL;GACA,QAAQ,cAAc,YAAY,WAAW,EAAE,eAAe,MAAM,CAAC,CAAC;GACvE;SACK;AACN,SAAO;GAAE;GAAW,QAAQ,EAAE;GAAE;;;;;;;;;AAUpC,SAAgB,kBAAkB,SAAmC;CACnE,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAAwB,EAAE;AAEhC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,WAAW,KAAK,KAAK,WAAW,OAAO,QAAQ;AACrD,OAAK,MAAM,QAAQ,YAAY,SAAS,EAAE;GACxC,MAAM,KAAK,KACR,SAAS,UAAU,KAAK,CACxB,QAAQ,eAAe,GAAG,CAC1B,MAAM,KAAK,IAAI,CACf,KAAK,IAAI;AACZ,OAAI,KAAK;IAAE;IAAM;IAAI;IAAO,CAAC;;;AAIjC,QAAO,IAAI,MACR,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,IAAI,EAAE,GAAG,cAAc,EAAE,GAAG,CACrE;;;;;;;;AASH,SAAgB,mBAAmB,SAAqC;CACtE,MAAM,OAAO,KAAK,KAAK,SAAS,kBAAkB;AAClD,QAAO,WAAW,KAAK,GAAG,OAAO;;;;;;;;AASnC,SAAgB,oBAAoB,SAAyC;CAC3E,MAAM,EAAE,WAAW,WAAW,WAAW,QAAQ;CACjD,MAAM,MAA8B,EAAE;AAEtC,MAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,OAAO,KAAK,KAAK,WAAW,OAAO,SAAS,kBAAkB;AACpE,MAAI,WAAW,KAAK,CAAE,KAAI,KAAK;GAAE;GAAM;GAAO,CAAC;;AAGjD,QAAO,IAAI,MAAM,GAAG,MAAM,EAAE,MAAM,cAAc,EAAE,MAAM,CAAC"}
@@ -2,13 +2,13 @@ import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth, re
2
2
  import { MlflowClient, PostResult, normalizeHost } from "../connectors/mlflow/client.js";
3
3
  import "../connectors/mlflow/index.js";
4
4
  import { DatasetRow, ReadEvalDatasetOptions, readEvalDataset, userTurns } from "./dataset.js";
5
- import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext } from "./types.js";
5
+ import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, EvalWebServer, MatchResult, Matcher, Severity, TestContext } from "./types.js";
6
6
  import { defineEval, defineEvalConfig } from "./define-eval.js";
7
- import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
7
+ import { DiscoveredEval, DiscoveredEvalConfig, discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./discover.js";
8
8
  import { HttpDriverOptions, createHttpDriver } from "./http-driver.js";
9
9
  import { JudgeConfig, JudgeScore, configureJudge, isJudgeConfigured } from "./judge.js";
10
10
  import { equals, includes, matches } from "./matchers.js";
11
11
  import { Assessment, ReportOutcome, buildAssessments, reportToMlflow } from "./mlflow-report.js";
12
12
  import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./report.js";
13
13
  import { RunEvalOptions, runEval } from "./run-eval.js";
14
- import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries } from "./run-evals.js";
14
+ import { EvalProgress, EvalRunSummary, RunEvalsOptions, loadRootEvalConfig, runEvalsInDir, runWithRetries } from "./run-evals.js";
@@ -2,13 +2,13 @@ import { resolveDatabricksAuth, resolveWorkspaceClient } from "../connectors/mlf
2
2
  import { MlflowClient, normalizeHost } from "../connectors/mlflow/client.js";
3
3
  import { readEvalDataset, userTurns } from "./dataset.js";
4
4
  import { defineEval, defineEvalConfig } from "./define-eval.js";
5
- import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
5
+ import { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./discover.js";
6
6
  import { createHttpDriver } from "./http-driver.js";
7
7
  import { configureJudge, isJudgeConfigured } from "./judge.js";
8
8
  import { equals, includes, matches } from "./matchers.js";
9
9
  import { buildAssessments, reportToMlflow } from "./mlflow-report.js";
10
10
  import { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatResultsJUnit, formatResultsJson, formatSummaryLine, summarize } from "./report.js";
11
11
  import { runEval } from "./run-eval.js";
12
- import { runEvalsInDir, runWithRetries } from "./run-evals.js";
12
+ import { loadRootEvalConfig, runEvalsInDir, runWithRetries } from "./run-evals.js";
13
13
 
14
14
  export { };
@@ -1,6 +1,6 @@
1
1
  import { WorkspaceClient } from "../workspace-client/index.js";
2
2
  import "./dataset.js";
3
- import { EvalResult } from "./types.js";
3
+ import { EvalConfig, EvalResult } from "./types.js";
4
4
  import { ReportOutcome } from "./mlflow-report.js";
5
5
  import { FinishOutcome } from "./mlflow-run.js";
6
6
 
@@ -98,6 +98,13 @@ interface EvalRunSummary {
98
98
  finish: FinishOutcome;
99
99
  };
100
100
  }
101
+ /**
102
+ * Load the root `evals.config.ts` under `rootDir` (the project root), or return
103
+ * `undefined` when there is none. This is the run-wide config carrying
104
+ * `baseUrl`/`webServer`; the CLI reads it to resolve options and manage the
105
+ * app-under-test lifecycle before calling {@link runEvalsInDir}.
106
+ */
107
+ declare function loadRootEvalConfig(rootDir: string): Promise<EvalConfig | undefined>;
101
108
  /**
102
109
  * Run `attempt` up to `1 + retries` times, stopping as soon as it returns a
103
110
  * result that is neither a thrown error / per-eval timeout (`error`) nor a
@@ -119,5 +126,5 @@ declare function runWithRetries(retries: number, attempt: (attemptNumber: number
119
126
  */
120
127
  declare function runEvalsInDir(options: RunEvalsOptions): Promise<EvalRunSummary>;
121
128
  //#endregion
122
- export { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir, runWithRetries };
129
+ export { EvalProgress, EvalRunSummary, RunEvalsOptions, loadRootEvalConfig, runEvalsInDir, runWithRetries };
123
130
  //# sourceMappingURL=run-evals.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAmBiB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAWU;EATV,MAAA;EAwDkB;;;;EAnDlB,IAAA;EALA;EAOA,MAAA;EAAA;EAEA,OAAA,GAAU,MAAA;EAAA;;;;;;EAOV,WAAA;EAiBA;;;;;EAXA,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UA6BF;IA3BE,cAAA;EAAA;EA6BS;;;AAGb;EA1BE,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EA4BnC;;;;EAvBJ,eAAA,GAAkB,eAAA;EAwB4B;EAtB9C,WAAA;EAuBoB;EArBpB,GAAA;EAqBwC;;;;AAE1C;EAjBE,SAAA;;;;;;EAMA,OAAA;EAYA;EAVA,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EA2OmC;EAzO5C,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;;;;;;;iBAuOrC,cAAA,CACpB,OAAA,UACA,OAAA,GAAU,aAAA,aAA0B,OAAA,CAAQ,UAAA,GAC5C,OAAA;EAAW,WAAA;AAAA,IACV,OAAA,CAAQ,UAAA;;;;;;iBA8GW,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
1
+ {"version":3,"file":"run-evals.d.ts","names":[],"sources":["../../src/evals/run-evals.ts"],"mappings":";;;;;;;UAoBiB,eAAA;;EAEf,OAAA;EAF8B;EAI9B,OAAA;EAWU;EATV,MAAA;EAwDkB;;;;EAnDlB,IAAA;EALA;EAOA,MAAA;EAAA;EAEA,OAAA,GAAU,MAAA;EAAA;;;;;;EAOV,WAAA;EAiBA;;;;;EAXA,MAAA;IACE,IAAA;IACA,KAAA;IACA,YAAA,UA6BF;IA3BE,cAAA;EAAA;EA6BS;;;AAGb;EA1BE,KAAA;IAAU,IAAA;IAAc,KAAA;IAAe,KAAA;EAAA;EA4BnC;;;;EAvBJ,eAAA,GAAkB,eAAA;EAwB4B;EAtB9C,WAAA;EAuBoB;EArBpB,GAAA;EAqBwC;;;;AAE1C;EAjBE,SAAA;;;;;;EAMA,OAAA;EAYA;EAVA,OAAA,IAAW,KAAA,EAAO,YAAA;AAAA;AAAA,KAGR,YAAA;EACN,IAAA;EAAoB,KAAA;AAAA;EACpB,IAAA;EAAqB,KAAA;AAAA;EACrB,IAAA;EAAe,EAAA;EAAY,KAAA;EAAe,KAAA;AAAA;EAC1C,IAAA;EAAgB,MAAA,EAAQ,UAAA;EAAY,KAAA;EAAe,KAAA;AAAA;AAAA,UAExC,cAAA;EACf,OAAA,EAAS,UAAA;EAyPmC;EAvP5C,MAAA;IAAW,KAAA;IAAe,MAAA,EAAQ,aAAA;IAAe,MAAA,EAAQ,aAAA;EAAA;AAAA;;;;;;;iBAmDrC,kBAAA,CACpB,OAAA,WACC,OAAA,CAAQ,UAAA;;;;;;;;;;;;iBAgMW,cAAA,CACpB,OAAA,UACA,OAAA,GAAU,aAAA,aAA0B,OAAA,CAAQ,UAAA,GAC5C,OAAA;EAAW,WAAA;AAAA,IACV,OAAA,CAAQ,UAAA;;;;;;iBA8GW,aAAA,CACpB,OAAA,EAAS,eAAA,GACR,OAAA,CAAQ,cAAA"}
@@ -1,6 +1,6 @@
1
1
  import { MlflowClient } from "../connectors/mlflow/client.js";
2
2
  import { readEvalDataset } from "./dataset.js";
3
- import { discoverEvalConfigs, discoverEvalFiles } from "./discover.js";
3
+ import { discoverEvalConfigs, discoverEvalFiles, findRootEvalConfig } from "./discover.js";
4
4
  import { createHttpDriver } from "./http-driver.js";
5
5
  import { configureJudge, teardownJudge } from "./judge.js";
6
6
  import { mapPool } from "./pool.js";
@@ -43,6 +43,17 @@ async function loadEvalConfig(file) {
43
43
  return resolveConfigDefault(await tsImportFile(file));
44
44
  }
45
45
  /**
46
+ * Load the root `evals.config.ts` under `rootDir` (the project root), or return
47
+ * `undefined` when there is none. This is the run-wide config carrying
48
+ * `baseUrl`/`webServer`; the CLI reads it to resolve options and manage the
49
+ * app-under-test lifecycle before calling {@link runEvalsInDir}.
50
+ */
51
+ async function loadRootEvalConfig(rootDir) {
52
+ const file = findRootEvalConfig(rootDir);
53
+ if (!file) return void 0;
54
+ return loadEvalConfig(file);
55
+ }
56
+ /**
46
57
  * Unwrap the config default export across module-interop shapes (see
47
58
  * {@link resolveEvalDefault}). A config has no `.test`, so the first plain
48
59
  * object reached through the `default` chain is taken as the config.
@@ -355,5 +366,5 @@ async function runEvalsInDir(options) {
355
366
  }
356
367
 
357
368
  //#endregion
358
- export { runEvalsInDir, runWithRetries };
369
+ export { loadRootEvalConfig, runEvalsInDir, runWithRetries };
359
370
  //# sourceMappingURL=run-evals.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"run-evals.js","names":["sleep"],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { setTimeout as sleep } from \"node:timers/promises\";\nimport { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport {\n type DiscoveredEval,\n discoverEvalConfigs,\n discoverEvalFiles,\n} from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalConfig, EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /**\n * Only run evals whose `tags` intersect this list. Empty/undefined runs all.\n * Tags live on the eval def, so filtering happens after each file is loaded.\n */\n tags?: string[];\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /**\n * Default per-eval timeout (ms): `runEval` races the whole test against it and\n * it also caps each driver turn. A per-eval `def.timeoutMs` overrides it, and\n * it wins over an agent's `evals.config.ts` `timeoutMs`. Unbounded when unset.\n */\n timeoutMs?: number;\n /**\n * Re-run an eval up to this many extra times when it fails on infrastructure —\n * a thrown error/timeout (`result.error`) or a transport/agent turn failure\n * (`result.infraFailure`). Assertion failures are never retried. Defaults to `0`.\n */\n retries?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Import a TypeScript file with tsx's programmatic loader so eval files run\n * without a build step. The specifier is indirected so the type checker doesn't\n * try to resolve tsx's internal entry.\n */\nasync function tsImportFile(file: string): Promise<unknown> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n return tsImport(pathToFileURL(file).href, import.meta.url);\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const mod = await tsImportFile(file);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Load an `evals.config.ts` file and return its default-exported\n * {@link EvalConfig}. A malformed/missing default surfaces as `undefined` so a\n * bad config never aborts a whole run.\n */\nasync function loadEvalConfig(file: string): Promise<EvalConfig | undefined> {\n const mod = await tsImportFile(file);\n return resolveConfigDefault(mod);\n}\n\n/**\n * Unwrap the config default export across module-interop shapes (see\n * {@link resolveEvalDefault}). A config has no `.test`, so the first plain\n * object reached through the `default` chain is taken as the config.\n */\nexport function resolveConfigDefault(mod: unknown): EvalConfig | undefined {\n let candidate: unknown = mod;\n // `i < 4` bounds the chain; no visited-set needed (cf. resolveEvalDefault).\n for (let i = 0; i < 4 && candidate; i++) {\n const next = (candidate as { default?: unknown }).default;\n if (next === undefined) {\n return typeof candidate === \"object\"\n ? (candidate as EvalConfig)\n : undefined;\n }\n candidate = next;\n }\n return undefined;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // Each attempt builds a fresh driver, so a retry never inherits the failed\n // attempt's thread. (runWithRetries defines what counts as retryable.)\n return await runWithRetries(options.retries ?? 0, () =>\n runEval(def, {\n id,\n driver: createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n // Cap the driver turn at the eval's effective timeout (runEval's signal also aborts it).\n timeoutMs: def.timeoutMs ?? options.timeoutMs,\n }),\n strict: options.strict,\n row,\n timeoutMs: options.timeoutMs,\n }),\n );\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Run one already-loaded eval (from the `loaded` pre-pass), expanding a\n * dataset-driven eval into one run per row. Appends one result per row to\n * `results`, emitting `start`/`result` around each. Never throws: a load error\n * (carried in `loadError`) or a dataset-read failure surfaces as a non-passing\n * result. `total` counts eval files, not rows — per-row detail is carried in the\n * result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n def: EvalDefinition,\n loadError: string | undefined,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n // Load failed in the pre-pass (def is a placeholder) → one non-passing result.\n if (loadError) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: loadError,\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Base delay (ms) before the first retry; doubled per attempt, full-jittered, capped. */\nconst DEFAULT_RETRY_BASE_DELAY_MS = 250;\n/** Ceiling for a single retry backoff wait (ms). */\nconst MAX_RETRY_DELAY_MS = 5_000;\n\n/**\n * Run `attempt` up to `1 + retries` times, stopping as soon as it returns a\n * result that is neither a thrown error / per-eval timeout (`error`) nor a\n * transport/agent turn failure (`infraFailure`). Assertion failures set\n * neither, so a failed-but-completed eval is returned on the first try and\n * never retried. Returns the last result when every attempt failed on infra.\n *\n * Between attempts it waits a full-jittered exponential backoff (infra flakes\n * are overload-correlated). `retries` is coerced to a finite non-negative\n * integer; `baseDelayMs: 0` disables the wait (tests).\n */\nexport async function runWithRetries(\n retries: number,\n attempt: (attemptNumber: number) => Promise<EvalResult>,\n options: { baseDelayMs?: number } = {},\n): Promise<EvalResult> {\n const baseDelayMs = options.baseDelayMs ?? DEFAULT_RETRY_BASE_DELAY_MS;\n const maxRetries = Number.isFinite(retries)\n ? Math.max(0, Math.floor(retries))\n : 0;\n const maxAttempts = 1 + maxRetries;\n let result: EvalResult;\n for (let n = 1; ; n++) {\n result = await attempt(n);\n const infraFailed = result.error !== undefined || result.infraFailure;\n if (!infraFailed || n >= maxAttempts) return result;\n if (baseDelayMs > 0) {\n // Full jitter: a random wait in [0, min(cap, base * 2^(n-1))].\n const ceiling = Math.min(baseDelayMs * 2 ** (n - 1), MAX_RETRY_DELAY_MS);\n await sleep(Math.random() * ceiling);\n }\n }\n}\n\n/**\n * Whether an eval's `tags` satisfy a `--tag` filter: `true` when the filter is\n * empty/undefined (no filtering), otherwise only when the eval shares at least\n * one tag with it. An eval with no tags never matches a non-empty filter.\n */\nexport function matchesTags(\n defTags: string[] | undefined,\n filterTags: string[] | undefined,\n): boolean {\n if (!filterTags || filterTags.length === 0) return true;\n return defTags?.some((t) => filterTags.includes(t)) ?? false;\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Resolve the work-pool width: `--concurrency` wins; else the lowest\n * `maxConcurrency` any *participating* agent's `evals.config.ts` requests (all\n * evals share one per-user stream budget, so the most conservative ceiling\n * governs); else {@link DEFAULT_CONCURRENCY}.\n */\nexport function deriveConcurrency(\n activeAgents: Set<string>,\n configs: Map<string, EvalConfig>,\n cliConcurrency: number | undefined,\n): number {\n const configMin = [...configs.entries()]\n .filter(([agent]) => activeAgents.has(agent))\n .map(([, c]) => c.maxConcurrency)\n .filter((n): n is number => typeof n === \"number\")\n .reduce<number | undefined>(\n (min, n) => (min === undefined ? n : Math.min(min, n)),\n undefined,\n );\n return cliConcurrency ?? configMin ?? DEFAULT_CONCURRENCY;\n}\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n\n // Load each agent's `evals.config.ts` (best-effort, per-agent): its settings\n // apply only to that agent's evals. A malformed/missing config never aborts\n // the run — the agent just falls back to CLI options and built-in defaults.\n const configs = new Map<string, EvalConfig>();\n for (const c of discoverEvalConfigs(root)) {\n try {\n const cfg = await loadEvalConfig(c.file);\n if (cfg) configs.set(c.agent, cfg);\n } catch {\n // Ignore: fall back to CLI options / defaults for this agent.\n }\n }\n\n // Load each eval def and apply the `--tag` filter up front. Tags live on the\n // def, so a tag miss removes the eval entirely (like the substring filter\n // excludes files) rather than surfacing as a result. Load failures are kept\n // so a broken file still reports as a non-passing result.\n const loaded: Array<{\n d: DiscoveredEval;\n def: EvalDefinition;\n loadError?: string;\n }> = [];\n for (const d of discovered) {\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n loaded.push({\n d,\n // No def loaded; placeholder def is never run (error short-circuits).\n def: { test: () => {} },\n loadError: err instanceof Error ? err.message : String(err),\n });\n continue;\n }\n if (!matchesTags(def.tags, options.tags)) continue;\n loaded.push({ d, def });\n }\n\n // Pool width from participating agents' configs (see {@link deriveConcurrency}).\n const activeAgents = new Set(loaded.map((l) => l.d.agent));\n const concurrency = deriveConcurrency(\n activeAgents,\n configs,\n options.concurrency,\n );\n\n const total = loaded.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each loaded (tag-filtered) eval through the bounded pool — one in-flight\n // stream per eval, so the pool respects the server's per-user stream cap (see\n // mapPool/concurrency). A dataset eval expands into per-row runs that execute\n // serially within its slot; results preserve discovery order (mapPool writes\n // by index) and row order within each file. Per-agent timeout is folded into\n // the file's options (CLI wins over `evals.config.ts`; `def.timeoutMs` still\n // overrides, applied inside runEval). `total` counts eval files, not dataset\n // rows — per-row detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n loaded,\n concurrency,\n async ({ d, def, loadError }, index) => {\n const fileResults: EvalResult[] = [];\n const fileOptions: RunEvalsOptions = {\n ...options,\n timeoutMs: options.timeoutMs ?? configs.get(d.agent)?.timeoutMs,\n };\n await runDiscovered(\n d,\n def,\n loadError,\n index,\n total,\n runId,\n fileOptions,\n emit,\n fileResults,\n );\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAqGA,eAAe,aAAa,MAAgC;CAC1D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;AAEH,QAAO,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI;;;;;AAM5D,eAAe,SAAS,MAAuC;CAE7D,MAAM,MAAM,mBADA,MAAM,aAAa,KAAK,CACD;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;AAQT,eAAe,eAAe,MAA+C;AAE3E,QAAO,qBADK,MAAM,aAAa,KAAK,CACJ;;;;;;;AAQlC,SAAgB,qBAAqB,KAAsC;CACzE,IAAI,YAAqB;AAEzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;EACvC,MAAM,OAAQ,UAAoC;AAClD,MAAI,SAAS,OACX,QAAO,OAAO,cAAc,WACvB,YACD;AAEN,cAAY;;;;;;;;;AAWhB,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAGF,SAAO,MAAM,eAAe,QAAQ,WAAW,SAC7C,QAAQ,KAAK;GACX;GACA,QAAQ,iBAAiB;IACvB,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IAEb,WAAW,IAAI,aAAa,QAAQ;IACrC,CAAC;GACF,QAAQ,QAAQ;GAChB;GACA,WAAW,QAAQ;GACpB,CAAC,CACH;UACM,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;;AAYL,eAAe,cACb,GACA,KACA,WACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;AAG3B,KAAI,WAAW;AACb,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO;GACR;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,MAAM,8BAA8B;;AAEpC,MAAM,qBAAqB;;;;;;;;;;;;AAa3B,eAAsB,eACpB,SACA,SACA,UAAoC,EAAE,EACjB;CACrB,MAAM,cAAc,QAAQ,eAAe;CAI3C,MAAM,cAAc,KAHD,OAAO,SAAS,QAAQ,GACvC,KAAK,IAAI,GAAG,KAAK,MAAM,QAAQ,CAAC,GAChC;CAEJ,IAAI;AACJ,MAAK,IAAI,IAAI,IAAK,KAAK;AACrB,WAAS,MAAM,QAAQ,EAAE;AAEzB,MAAI,EADgB,OAAO,UAAU,UAAa,OAAO,iBACrC,KAAK,YAAa,QAAO;AAC7C,MAAI,cAAc,GAAG;GAEnB,MAAM,UAAU,KAAK,IAAI,cAAc,MAAM,IAAI,IAAI,mBAAmB;AACxE,SAAMA,WAAM,KAAK,QAAQ,GAAG,QAAQ;;;;;;;;;AAU1C,SAAgB,YACd,SACA,YACS;AACT,KAAI,CAAC,cAAc,WAAW,WAAW,EAAG,QAAO;AACnD,QAAO,SAAS,MAAM,MAAM,WAAW,SAAS,EAAE,CAAC,IAAI;;;AAIzD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;;AAQ5B,SAAgB,kBACd,cACA,SACA,gBACQ;CACR,MAAM,YAAY,CAAC,GAAG,QAAQ,SAAS,CAAC,CACrC,QAAQ,CAAC,WAAW,aAAa,IAAI,MAAM,CAAC,CAC5C,KAAK,GAAG,OAAO,EAAE,eAAe,CAChC,QAAQ,MAAmB,OAAO,MAAM,SAAS,CACjD,QACE,KAAK,MAAO,QAAQ,SAAY,IAAI,KAAK,IAAI,KAAK,EAAE,EACrD,OACD;AACH,QAAO,kBAAkB,aAAa;;;;;;;AAQxC,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CAKvC,MAAM,0BAAU,IAAI,KAAyB;AAC7C,MAAK,MAAM,KAAK,oBAAoB,KAAK,CACvC,KAAI;EACF,MAAM,MAAM,MAAM,eAAe,EAAE,KAAK;AACxC,MAAI,IAAK,SAAQ,IAAI,EAAE,OAAO,IAAI;SAC5B;CASV,MAAM,SAID,EAAE;AACP,MAAK,MAAM,KAAK,YAAY;EAC1B,IAAI;AACJ,MAAI;AACF,SAAM,MAAM,SAAS,EAAE,KAAK;WACrB,KAAK;AACZ,UAAO,KAAK;IACV;IAEA,KAAK,EAAE,YAAY,IAAI;IACvB,WAAW,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;IAC5D,CAAC;AACF;;AAEF,MAAI,CAAC,YAAY,IAAI,MAAM,QAAQ,KAAK,CAAE;AAC1C,SAAO,KAAK;GAAE;GAAG;GAAK,CAAC;;CAKzB,MAAM,cAAc,kBADC,IAAI,IAAI,OAAO,KAAK,MAAM,EAAE,EAAE,MAAM,CAAC,EAGxD,SACA,QAAQ,YACT;CAED,MAAM,QAAQ,OAAO;AACrB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkCtC,MAAM,WAvBU,MAAM,QACpB,QACA,aACA,OAAO,EAAE,GAAG,KAAK,aAAa,UAAU;GACtC,MAAM,cAA4B,EAAE;GACpC,MAAM,cAA+B;IACnC,GAAG;IACH,WAAW,QAAQ,aAAa,QAAQ,IAAI,EAAE,MAAM,EAAE;IACvD;AACD,SAAM,cACJ,GACA,KACA,WACA,OACA,OACA,OACA,aACA,MACA,YACD;AACD,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
1
+ {"version":3,"file":"run-evals.js","names":["sleep"],"sources":["../../src/evals/run-evals.ts"],"sourcesContent":["import { setTimeout as sleep } from \"node:timers/promises\";\nimport { pathToFileURL } from \"node:url\";\n\nimport { MlflowClient } from \"../connectors/mlflow\";\nimport type { WorkspaceClient } from \"../workspace-client\";\nimport { type DatasetRow, readEvalDataset } from \"./dataset\";\nimport {\n type DiscoveredEval,\n discoverEvalConfigs,\n discoverEvalFiles,\n findRootEvalConfig,\n} from \"./discover\";\nimport { createHttpDriver } from \"./http-driver\";\nimport { configureJudge, teardownJudge } from \"./judge\";\nimport { type ReportOutcome, reportToMlflow } from \"./mlflow-report\";\nimport { createEvalRun, type FinishOutcome, finishEvalRun } from \"./mlflow-run\";\nimport { mapPool } from \"./pool\";\nimport { runEval } from \"./run-eval\";\nimport type { EvalConfig, EvalDefinition, EvalResult } from \"./types\";\n\nexport interface RunEvalsOptions {\n /** Project root containing `server/agents/`. Defaults to `process.cwd()`. */\n rootDir?: string;\n /** Base URL of the running app to drive, e.g. `http://localhost:3000`. */\n baseUrl: string;\n /** Substring filter on `<agent>/<id>` (or an exact agent id). */\n filter?: string;\n /**\n * Only run evals whose `tags` intersect this list. Empty/undefined runs all.\n * Tags live on the eval def, so filtering happens after each file is loaded.\n */\n tags?: string[];\n /** Soft assertion failures also fail the eval. */\n strict?: boolean;\n /** Extra request headers for the driver (e.g. auth for a deployed app). */\n headers?: Record<string, string>;\n /**\n * Max evals to drive concurrently. Each eval opens one stream to the app as\n * the same user, so keep this at or below the app's\n * `maxConcurrentStreamsPerUser` (default 5) or the surplus streams hit the\n * 429 guard. Defaults to 4; clamped to `[1, total]`.\n */\n concurrency?: number;\n /**\n * When set, create a native MLflow \"Evaluation run\": each eval's trace is\n * linked to the run, pass/fail is written as feedback, and aggregate metrics\n * are logged. Requires Databricks creds + the target experiment.\n */\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n /** SQL warehouse id for writing assessments to UC-backed (V4) traces. */\n sqlWarehouseId?: string;\n };\n /**\n * When set, enable `t.judge.*` LLM-as-judge scoring via autoevals against a\n * Databricks serving endpoint (`model`).\n */\n judge?: { host: string; token: string; model: string };\n /**\n * Workspace client used to read managed evaluation datasets (for evals that\n * declare `dataset`). Required alongside {@link warehouseId} for those evals.\n */\n workspaceClient?: WorkspaceClient;\n /** SQL warehouse id used to read managed evaluation datasets. */\n warehouseId?: string;\n /** Wall-clock timestamp (ms) for run create/finish — pass `Date.now()`. */\n now?: number;\n /**\n * Default per-eval timeout (ms): `runEval` races the whole test against it and\n * it also caps each driver turn. A per-eval `def.timeoutMs` overrides it, and\n * it wins over an agent's `evals.config.ts` `timeoutMs`. Unbounded when unset.\n */\n timeoutMs?: number;\n /**\n * Re-run an eval up to this many extra times when it fails on infrastructure —\n * a thrown error/timeout (`result.error`) or a transport/agent turn failure\n * (`result.infraFailure`). Assertion failures are never retried. Defaults to `0`.\n */\n retries?: number;\n /** Progress callback, invoked as evals are discovered, started, and finished. */\n onEvent?: (event: EvalProgress) => void;\n}\n\nexport type EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: EvalResult; index: number; total: number };\n\nexport interface EvalRunSummary {\n results: EvalResult[];\n /** Present when an MLflow evaluation run was created. */\n mlflow?: { runId: string; report: ReportOutcome; finish: FinishOutcome };\n}\n\n/**\n * Import a TypeScript file with tsx's programmatic loader so eval files run\n * without a build step. The specifier is indirected so the type checker doesn't\n * try to resolve tsx's internal entry.\n */\nasync function tsImportFile(file: string): Promise<unknown> {\n const tsxApi = \"tsx/esm/api\";\n let tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n try {\n ({ tsImport } = (await import(tsxApi)) as {\n tsImport: (specifier: string, parentURL: string) => Promise<unknown>;\n });\n } catch {\n throw new Error(\n \"Running .eval.ts files requires `tsx`. Install it as a dev dependency (`pnpm add -D tsx`).\",\n );\n }\n return tsImport(pathToFileURL(file).href, import.meta.url);\n}\n\n/**\n * Load a `*.eval.ts` file and return its default-exported {@link EvalDefinition}.\n */\nasync function loadEval(file: string): Promise<EvalDefinition> {\n const mod = await tsImportFile(file);\n const def = resolveEvalDefault(mod);\n if (!def) {\n throw new Error(`${file}: must default-export defineEval({ test })`);\n }\n return def;\n}\n\n/**\n * Load an `evals.config.ts` file and return its default-exported\n * {@link EvalConfig}. A malformed/missing default surfaces as `undefined` so a\n * bad config never aborts a whole run.\n */\nasync function loadEvalConfig(file: string): Promise<EvalConfig | undefined> {\n const mod = await tsImportFile(file);\n return resolveConfigDefault(mod);\n}\n\n/**\n * Load the root `evals.config.ts` under `rootDir` (the project root), or return\n * `undefined` when there is none. This is the run-wide config carrying\n * `baseUrl`/`webServer`; the CLI reads it to resolve options and manage the\n * app-under-test lifecycle before calling {@link runEvalsInDir}.\n */\nexport async function loadRootEvalConfig(\n rootDir: string,\n): Promise<EvalConfig | undefined> {\n const file = findRootEvalConfig(rootDir);\n if (!file) return undefined;\n return loadEvalConfig(file);\n}\n\n/**\n * Unwrap the config default export across module-interop shapes (see\n * {@link resolveEvalDefault}). A config has no `.test`, so the first plain\n * object reached through the `default` chain is taken as the config.\n */\nexport function resolveConfigDefault(mod: unknown): EvalConfig | undefined {\n let candidate: unknown = mod;\n // `i < 4` bounds the chain; no visited-set needed (cf. resolveEvalDefault).\n for (let i = 0; i < 4 && candidate; i++) {\n const next = (candidate as { default?: unknown }).default;\n if (next === undefined) {\n return typeof candidate === \"object\"\n ? (candidate as EvalConfig)\n : undefined;\n }\n candidate = next;\n }\n return undefined;\n}\n\n/**\n * Unwrap the eval default export across module-interop shapes. Depending on\n * whether the eval file is treated as ESM or CJS, the value lands at\n * `mod.default` (ESM), `mod.default.default` (CJS `__esModule` double-wrap), or\n * `mod` itself. Returns the first candidate that looks like an eval.\n */\nexport function resolveEvalDefault(mod: unknown): EvalDefinition | undefined {\n let candidate: unknown = mod;\n for (let i = 0; i < 4 && candidate; i++) {\n if (typeof (candidate as EvalDefinition).test === \"function\") {\n return candidate as EvalDefinition;\n }\n candidate = (candidate as { default?: unknown }).default;\n }\n return undefined;\n}\n\n/**\n * Run one eval turn against a fresh driver. Never throws — a run failure becomes\n * a non-passing {@link EvalResult} so one bad eval can't abort the run. `row`\n * binds the current managed-dataset row (see {@link resolveDatasetRows}), or is\n * `undefined` for a plain single-run eval.\n */\nasync function runOne(\n d: DiscoveredEval,\n id: string,\n def: EvalDefinition,\n row: DatasetRow | undefined,\n runId: string | undefined,\n options: RunEvalsOptions,\n): Promise<EvalResult> {\n try {\n // Each attempt builds a fresh driver, so a retry never inherits the failed\n // attempt's thread. (runWithRetries defines what counts as retryable.)\n return await runWithRetries(options.retries ?? 0, () =>\n runEval(def, {\n id,\n driver: createHttpDriver({\n baseUrl: options.baseUrl,\n agent: def.agent ?? d.agent,\n headers: options.headers,\n mlflowRunId: runId,\n // Cap the driver turn at the eval's effective timeout (runEval's signal also aborts it).\n timeoutMs: def.timeoutMs ?? options.timeoutMs,\n }),\n strict: options.strict,\n row,\n timeoutMs: options.timeoutMs,\n }),\n );\n } catch (err) {\n return {\n id,\n assertions: [],\n passed: false,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Resolve the rows a (possibly dataset-driven) eval runs over. A plain eval\n * yields a single `undefined` row; a dataset eval reads its Unity Catalog table\n * via {@link readEvalDataset}. On misconfiguration or read failure, returns a\n * single `undefined` row plus an `error`, so the eval still surfaces one result.\n */\nexport async function resolveDatasetRows(\n def: EvalDefinition,\n options: RunEvalsOptions,\n): Promise<{ rows: Array<DatasetRow | undefined>; error?: string }> {\n if (!def.dataset) return { rows: [undefined] };\n if (!options.workspaceClient || !options.warehouseId) {\n return {\n rows: [undefined],\n error:\n \"dataset eval requires a workspace client and warehouse (pass --warehouse-id)\",\n };\n }\n try {\n const rows = await readEvalDataset(options.workspaceClient, {\n table: def.dataset.table,\n warehouseId: options.warehouseId,\n limit: def.dataset.limit,\n });\n if (rows.length === 0) {\n return {\n rows: [undefined],\n error: `dataset \"${def.dataset.table}\" returned no rows`,\n };\n }\n return { rows };\n } catch (err) {\n return {\n rows: [undefined],\n error: err instanceof Error ? err.message : String(err),\n };\n }\n}\n\n/**\n * Run one already-loaded eval (from the `loaded` pre-pass), expanding a\n * dataset-driven eval into one run per row. Appends one result per row to\n * `results`, emitting `start`/`result` around each. Never throws: a load error\n * (carried in `loadError`) or a dataset-read failure surfaces as a non-passing\n * result. `total` counts eval files, not rows — per-row detail is carried in the\n * result id (`[row i/n]`).\n */\nasync function runDiscovered(\n d: DiscoveredEval,\n def: EvalDefinition,\n loadError: string | undefined,\n index: number,\n total: number,\n runId: string | undefined,\n options: RunEvalsOptions,\n emit: (event: EvalProgress) => void,\n results: EvalResult[],\n): Promise<void> {\n const id = `${d.agent}/${d.id}`;\n\n // Load failed in the pre-pass (def is a placeholder) → one non-passing result.\n if (loadError) {\n emit({ type: \"start\", id, index, total });\n const result: EvalResult = {\n id,\n assertions: [],\n passed: false,\n error: loadError,\n };\n results.push(result);\n emit({ type: \"result\", result, index, total });\n return;\n }\n\n const { rows, error: datasetError } = await resolveDatasetRows(def, options);\n\n for (let r = 0; r < rows.length; r++) {\n const rowId =\n def.dataset && rows.length > 1\n ? `${id} [row ${r + 1}/${rows.length}]`\n : id;\n emit({ type: \"start\", id: rowId, index, total });\n const result: EvalResult = datasetError\n ? { id: rowId, assertions: [], passed: false, error: datasetError }\n : await runOne(d, rowId, def, rows[r], runId, options);\n results.push(result);\n emit({ type: \"result\", result, index, total });\n }\n}\n\n/** Base delay (ms) before the first retry; doubled per attempt, full-jittered, capped. */\nconst DEFAULT_RETRY_BASE_DELAY_MS = 250;\n/** Ceiling for a single retry backoff wait (ms). */\nconst MAX_RETRY_DELAY_MS = 5_000;\n\n/**\n * Run `attempt` up to `1 + retries` times, stopping as soon as it returns a\n * result that is neither a thrown error / per-eval timeout (`error`) nor a\n * transport/agent turn failure (`infraFailure`). Assertion failures set\n * neither, so a failed-but-completed eval is returned on the first try and\n * never retried. Returns the last result when every attempt failed on infra.\n *\n * Between attempts it waits a full-jittered exponential backoff (infra flakes\n * are overload-correlated). `retries` is coerced to a finite non-negative\n * integer; `baseDelayMs: 0` disables the wait (tests).\n */\nexport async function runWithRetries(\n retries: number,\n attempt: (attemptNumber: number) => Promise<EvalResult>,\n options: { baseDelayMs?: number } = {},\n): Promise<EvalResult> {\n const baseDelayMs = options.baseDelayMs ?? DEFAULT_RETRY_BASE_DELAY_MS;\n const maxRetries = Number.isFinite(retries)\n ? Math.max(0, Math.floor(retries))\n : 0;\n const maxAttempts = 1 + maxRetries;\n let result: EvalResult;\n for (let n = 1; ; n++) {\n result = await attempt(n);\n const infraFailed = result.error !== undefined || result.infraFailure;\n if (!infraFailed || n >= maxAttempts) return result;\n if (baseDelayMs > 0) {\n // Full jitter: a random wait in [0, min(cap, base * 2^(n-1))].\n const ceiling = Math.min(baseDelayMs * 2 ** (n - 1), MAX_RETRY_DELAY_MS);\n await sleep(Math.random() * ceiling);\n }\n }\n}\n\n/**\n * Whether an eval's `tags` satisfy a `--tag` filter: `true` when the filter is\n * empty/undefined (no filtering), otherwise only when the eval shares at least\n * one tag with it. An eval with no tags never matches a non-empty filter.\n */\nexport function matchesTags(\n defTags: string[] | undefined,\n filterTags: string[] | undefined,\n): boolean {\n if (!filterTags || filterTags.length === 0) return true;\n return defTags?.some((t) => filterTags.includes(t)) ?? false;\n}\n\n/** Configure the LLM judge when judge creds were supplied; otherwise a no-op. */\nasync function maybeConfigureJudge(options: RunEvalsOptions): Promise<void> {\n if (!options.judge) return;\n await configureJudge({\n client: new MlflowClient(options.judge.host, options.judge.token),\n token: options.judge.token,\n model: options.judge.model,\n });\n}\n\n/**\n * Report per-eval assessments and finish the MLflow run, when one was created.\n * Returns the run summary, or `undefined` when there was no run to finalize.\n */\nasync function finalizeMlflow(\n client: MlflowClient | undefined,\n runId: string | undefined,\n results: EvalResult[],\n options: RunEvalsOptions,\n): Promise<EvalRunSummary[\"mlflow\"]> {\n if (!client || !runId) return undefined;\n // reportToMlflow is not supposed to throw, but if it ever does the run must\n // still be finished — otherwise it hangs in RUNNING forever.\n let report: ReportOutcome = { written: 0, skipped: 0, failures: [] };\n try {\n report = await reportToMlflow(\n client,\n results,\n options.mlflow?.sqlWarehouseId,\n );\n } catch (err) {\n report.failures.push({\n traceId: \"(report)\",\n error: err instanceof Error ? err.message : String(err),\n });\n }\n const finish = await finishEvalRun(client, {\n runId,\n results,\n endTime: options.now ?? Date.now(),\n });\n return { runId, report, finish };\n}\n\n/**\n * Default max evals in flight. Each eval opens one stream to the app as the\n * same user; the server caps concurrent streams per user at 5 by default\n * (`maxConcurrentStreamsPerUser`), so 4 leaves headroom under that limit.\n */\nconst DEFAULT_CONCURRENCY = 4;\n\n/**\n * Resolve the work-pool width: `--concurrency` wins; else the lowest\n * `maxConcurrency` any *participating* agent's `evals.config.ts` requests (all\n * evals share one per-user stream budget, so the most conservative ceiling\n * governs); else {@link DEFAULT_CONCURRENCY}.\n */\nexport function deriveConcurrency(\n activeAgents: Set<string>,\n configs: Map<string, EvalConfig>,\n cliConcurrency: number | undefined,\n): number {\n const configMin = [...configs.entries()]\n .filter(([agent]) => activeAgents.has(agent))\n .map(([, c]) => c.maxConcurrency)\n .filter((n): n is number => typeof n === \"number\")\n .reduce<number | undefined>(\n (min, n) => (min === undefined ? n : Math.min(min, n)),\n undefined,\n );\n return cliConcurrency ?? configMin ?? DEFAULT_CONCURRENCY;\n}\n\n/**\n * Discover, load, and run every eval under each agent's `evals/` dir, driving\n * the agents on a running app. Never throws for an individual eval — load/run\n * failures become non-passing {@link EvalResult}s.\n */\nexport async function runEvalsInDir(\n options: RunEvalsOptions,\n): Promise<EvalRunSummary> {\n const root = options.rootDir ?? process.cwd();\n const now = options.now ?? Date.now();\n let discovered = discoverEvalFiles(root);\n\n if (options.filter) {\n const f = options.filter;\n discovered = discovered.filter(\n (d) => d.agent === f || `${d.agent}/${d.id}`.includes(f),\n );\n }\n\n const emit = options.onEvent ?? (() => {});\n\n // Load each agent's `evals.config.ts` (best-effort, per-agent): its settings\n // apply only to that agent's evals. A malformed/missing config never aborts\n // the run — the agent just falls back to CLI options and built-in defaults.\n const configs = new Map<string, EvalConfig>();\n for (const c of discoverEvalConfigs(root)) {\n try {\n const cfg = await loadEvalConfig(c.file);\n if (cfg) configs.set(c.agent, cfg);\n } catch {\n // Ignore: fall back to CLI options / defaults for this agent.\n }\n }\n\n // Load each eval def and apply the `--tag` filter up front. Tags live on the\n // def, so a tag miss removes the eval entirely (like the substring filter\n // excludes files) rather than surfacing as a result. Load failures are kept\n // so a broken file still reports as a non-passing result.\n const loaded: Array<{\n d: DiscoveredEval;\n def: EvalDefinition;\n loadError?: string;\n }> = [];\n for (const d of discovered) {\n let def: EvalDefinition;\n try {\n def = await loadEval(d.file);\n } catch (err) {\n loaded.push({\n d,\n // No def loaded; placeholder def is never run (error short-circuits).\n def: { test: () => {} },\n loadError: err instanceof Error ? err.message : String(err),\n });\n continue;\n }\n if (!matchesTags(def.tags, options.tags)) continue;\n loaded.push({ d, def });\n }\n\n // Pool width from participating agents' configs (see {@link deriveConcurrency}).\n const activeAgents = new Set(loaded.map((l) => l.d.agent));\n const concurrency = deriveConcurrency(\n activeAgents,\n configs,\n options.concurrency,\n );\n\n const total = loaded.length;\n emit({ type: \"discovered\", total });\n\n // The judge sets OPENAI_* env vars globally (autoevals reads them per call),\n // so tear them down in `finally` once the run is over — pass or throw — so\n // the bearer doesn't linger in process.env.\n await maybeConfigureJudge(options);\n try {\n // Create the MLflow evaluation run up front so each eval's trace can be\n // linked to it as it runs. One client is shared by run create/finish and\n // the per-trace assessment writes.\n let runId: string | undefined;\n let mlflowClient: MlflowClient | undefined;\n if (options.mlflow) {\n mlflowClient = new MlflowClient(\n options.mlflow.host,\n options.mlflow.token,\n );\n runId = await createEvalRun(mlflowClient, {\n experimentId: options.mlflow.experimentId,\n runName: `appkit-eval ${new Date(now).toISOString()}`,\n startTime: now,\n });\n emit({ type: \"run-created\", runId });\n }\n\n // Run each loaded (tag-filtered) eval through the bounded pool — one in-flight\n // stream per eval, so the pool respects the server's per-user stream cap (see\n // mapPool/concurrency). A dataset eval expands into per-row runs that execute\n // serially within its slot; results preserve discovery order (mapPool writes\n // by index) and row order within each file. Per-agent timeout is folded into\n // the file's options (CLI wins over `evals.config.ts`; `def.timeoutMs` still\n // overrides, applied inside runEval). `total` counts eval files, not dataset\n // rows — per-row detail is carried in the result id (`[row i/n]`).\n const perFile = await mapPool(\n loaded,\n concurrency,\n async ({ d, def, loadError }, index) => {\n const fileResults: EvalResult[] = [];\n const fileOptions: RunEvalsOptions = {\n ...options,\n timeoutMs: options.timeoutMs ?? configs.get(d.agent)?.timeoutMs,\n };\n await runDiscovered(\n d,\n def,\n loadError,\n index,\n total,\n runId,\n fileOptions,\n emit,\n fileResults,\n );\n return fileResults;\n },\n );\n const results = perFile.flat();\n\n const summary: EvalRunSummary = { results };\n const mlflow = await finalizeMlflow(mlflowClient, runId, results, options);\n if (mlflow) summary.mlflow = mlflow;\n return summary;\n } finally {\n teardownJudge();\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;AAsGA,eAAe,aAAa,MAAgC;CAC1D,MAAM,SAAS;CACf,IAAI;AACJ,KAAI;AACF,GAAC,CAAE,YAAc,MAAM,OAAO;SAGxB;AACN,QAAM,IAAI,MACR,6FACD;;AAEH,QAAO,SAAS,cAAc,KAAK,CAAC,MAAM,OAAO,KAAK,IAAI;;;;;AAM5D,eAAe,SAAS,MAAuC;CAE7D,MAAM,MAAM,mBADA,MAAM,aAAa,KAAK,CACD;AACnC,KAAI,CAAC,IACH,OAAM,IAAI,MAAM,GAAG,KAAK,4CAA4C;AAEtE,QAAO;;;;;;;AAQT,eAAe,eAAe,MAA+C;AAE3E,QAAO,qBADK,MAAM,aAAa,KAAK,CACJ;;;;;;;;AASlC,eAAsB,mBACpB,SACiC;CACjC,MAAM,OAAO,mBAAmB,QAAQ;AACxC,KAAI,CAAC,KAAM,QAAO;AAClB,QAAO,eAAe,KAAK;;;;;;;AAQ7B,SAAgB,qBAAqB,KAAsC;CACzE,IAAI,YAAqB;AAEzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;EACvC,MAAM,OAAQ,UAAoC;AAClD,MAAI,SAAS,OACX,QAAO,OAAO,cAAc,WACvB,YACD;AAEN,cAAY;;;;;;;;;AAWhB,SAAgB,mBAAmB,KAA0C;CAC3E,IAAI,YAAqB;AACzB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,WAAW,KAAK;AACvC,MAAI,OAAQ,UAA6B,SAAS,WAChD,QAAO;AAET,cAAa,UAAoC;;;;;;;;;AAWrD,eAAe,OACb,GACA,IACA,KACA,KACA,OACA,SACqB;AACrB,KAAI;AAGF,SAAO,MAAM,eAAe,QAAQ,WAAW,SAC7C,QAAQ,KAAK;GACX;GACA,QAAQ,iBAAiB;IACvB,SAAS,QAAQ;IACjB,OAAO,IAAI,SAAS,EAAE;IACtB,SAAS,QAAQ;IACjB,aAAa;IAEb,WAAW,IAAI,aAAa,QAAQ;IACrC,CAAC;GACF,QAAQ,QAAQ;GAChB;GACA,WAAW,QAAQ;GACpB,CAAC,CACH;UACM,KAAK;AACZ,SAAO;GACL;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;AAUL,eAAsB,mBACpB,KACA,SACkE;AAClE,KAAI,CAAC,IAAI,QAAS,QAAO,EAAE,MAAM,CAAC,OAAU,EAAE;AAC9C,KAAI,CAAC,QAAQ,mBAAmB,CAAC,QAAQ,YACvC,QAAO;EACL,MAAM,CAAC,OAAU;EACjB,OACE;EACH;AAEH,KAAI;EACF,MAAM,OAAO,MAAM,gBAAgB,QAAQ,iBAAiB;GAC1D,OAAO,IAAI,QAAQ;GACnB,aAAa,QAAQ;GACrB,OAAO,IAAI,QAAQ;GACpB,CAAC;AACF,MAAI,KAAK,WAAW,EAClB,QAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,YAAY,IAAI,QAAQ,MAAM;GACtC;AAEH,SAAO,EAAE,MAAM;UACR,KAAK;AACZ,SAAO;GACL,MAAM,CAAC,OAAU;GACjB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD;;;;;;;;;;;AAYL,eAAe,cACb,GACA,KACA,WACA,OACA,OACA,OACA,SACA,MACA,SACe;CACf,MAAM,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE;AAG3B,KAAI,WAAW;AACb,OAAK;GAAE,MAAM;GAAS;GAAI;GAAO;GAAO,CAAC;EACzC,MAAM,SAAqB;GACzB;GACA,YAAY,EAAE;GACd,QAAQ;GACR,OAAO;GACR;AACD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;AAC9C;;CAGF,MAAM,EAAE,MAAM,OAAO,iBAAiB,MAAM,mBAAmB,KAAK,QAAQ;AAE5E,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,QACJ,IAAI,WAAW,KAAK,SAAS,IACzB,GAAG,GAAG,QAAQ,IAAI,EAAE,GAAG,KAAK,OAAO,KACnC;AACN,OAAK;GAAE,MAAM;GAAS,IAAI;GAAO;GAAO;GAAO,CAAC;EAChD,MAAM,SAAqB,eACvB;GAAE,IAAI;GAAO,YAAY,EAAE;GAAE,QAAQ;GAAO,OAAO;GAAc,GACjE,MAAM,OAAO,GAAG,OAAO,KAAK,KAAK,IAAI,OAAO,QAAQ;AACxD,UAAQ,KAAK,OAAO;AACpB,OAAK;GAAE,MAAM;GAAU;GAAQ;GAAO;GAAO,CAAC;;;;AAKlD,MAAM,8BAA8B;;AAEpC,MAAM,qBAAqB;;;;;;;;;;;;AAa3B,eAAsB,eACpB,SACA,SACA,UAAoC,EAAE,EACjB;CACrB,MAAM,cAAc,QAAQ,eAAe;CAI3C,MAAM,cAAc,KAHD,OAAO,SAAS,QAAQ,GACvC,KAAK,IAAI,GAAG,KAAK,MAAM,QAAQ,CAAC,GAChC;CAEJ,IAAI;AACJ,MAAK,IAAI,IAAI,IAAK,KAAK;AACrB,WAAS,MAAM,QAAQ,EAAE;AAEzB,MAAI,EADgB,OAAO,UAAU,UAAa,OAAO,iBACrC,KAAK,YAAa,QAAO;AAC7C,MAAI,cAAc,GAAG;GAEnB,MAAM,UAAU,KAAK,IAAI,cAAc,MAAM,IAAI,IAAI,mBAAmB;AACxE,SAAMA,WAAM,KAAK,QAAQ,GAAG,QAAQ;;;;;;;;;AAU1C,SAAgB,YACd,SACA,YACS;AACT,KAAI,CAAC,cAAc,WAAW,WAAW,EAAG,QAAO;AACnD,QAAO,SAAS,MAAM,MAAM,WAAW,SAAS,EAAE,CAAC,IAAI;;;AAIzD,eAAe,oBAAoB,SAAyC;AAC1E,KAAI,CAAC,QAAQ,MAAO;AACpB,OAAM,eAAe;EACnB,QAAQ,IAAI,aAAa,QAAQ,MAAM,MAAM,QAAQ,MAAM,MAAM;EACjE,OAAO,QAAQ,MAAM;EACrB,OAAO,QAAQ,MAAM;EACtB,CAAC;;;;;;AAOJ,eAAe,eACb,QACA,OACA,SACA,SACmC;AACnC,KAAI,CAAC,UAAU,CAAC,MAAO,QAAO;CAG9B,IAAI,SAAwB;EAAE,SAAS;EAAG,SAAS;EAAG,UAAU,EAAE;EAAE;AACpE,KAAI;AACF,WAAS,MAAM,eACb,QACA,SACA,QAAQ,QAAQ,eACjB;UACM,KAAK;AACZ,SAAO,SAAS,KAAK;GACnB,SAAS;GACT,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;GACxD,CAAC;;CAEJ,MAAM,SAAS,MAAM,cAAc,QAAQ;EACzC;EACA;EACA,SAAS,QAAQ,OAAO,KAAK,KAAK;EACnC,CAAC;AACF,QAAO;EAAE;EAAO;EAAQ;EAAQ;;;;;;;AAQlC,MAAM,sBAAsB;;;;;;;AAQ5B,SAAgB,kBACd,cACA,SACA,gBACQ;CACR,MAAM,YAAY,CAAC,GAAG,QAAQ,SAAS,CAAC,CACrC,QAAQ,CAAC,WAAW,aAAa,IAAI,MAAM,CAAC,CAC5C,KAAK,GAAG,OAAO,EAAE,eAAe,CAChC,QAAQ,MAAmB,OAAO,MAAM,SAAS,CACjD,QACE,KAAK,MAAO,QAAQ,SAAY,IAAI,KAAK,IAAI,KAAK,EAAE,EACrD,OACD;AACH,QAAO,kBAAkB,aAAa;;;;;;;AAQxC,eAAsB,cACpB,SACyB;CACzB,MAAM,OAAO,QAAQ,WAAW,QAAQ,KAAK;CAC7C,MAAM,MAAM,QAAQ,OAAO,KAAK,KAAK;CACrC,IAAI,aAAa,kBAAkB,KAAK;AAExC,KAAI,QAAQ,QAAQ;EAClB,MAAM,IAAI,QAAQ;AAClB,eAAa,WAAW,QACrB,MAAM,EAAE,UAAU,KAAK,GAAG,EAAE,MAAM,GAAG,EAAE,KAAK,SAAS,EAAE,CACzD;;CAGH,MAAM,OAAO,QAAQ,kBAAkB;CAKvC,MAAM,0BAAU,IAAI,KAAyB;AAC7C,MAAK,MAAM,KAAK,oBAAoB,KAAK,CACvC,KAAI;EACF,MAAM,MAAM,MAAM,eAAe,EAAE,KAAK;AACxC,MAAI,IAAK,SAAQ,IAAI,EAAE,OAAO,IAAI;SAC5B;CASV,MAAM,SAID,EAAE;AACP,MAAK,MAAM,KAAK,YAAY;EAC1B,IAAI;AACJ,MAAI;AACF,SAAM,MAAM,SAAS,EAAE,KAAK;WACrB,KAAK;AACZ,UAAO,KAAK;IACV;IAEA,KAAK,EAAE,YAAY,IAAI;IACvB,WAAW,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI;IAC5D,CAAC;AACF;;AAEF,MAAI,CAAC,YAAY,IAAI,MAAM,QAAQ,KAAK,CAAE;AAC1C,SAAO,KAAK;GAAE;GAAG;GAAK,CAAC;;CAKzB,MAAM,cAAc,kBADC,IAAI,IAAI,OAAO,KAAK,MAAM,EAAE,EAAE,MAAM,CAAC,EAGxD,SACA,QAAQ,YACT;CAED,MAAM,QAAQ,OAAO;AACrB,MAAK;EAAE,MAAM;EAAc;EAAO,CAAC;AAKnC,OAAM,oBAAoB,QAAQ;AAClC,KAAI;EAIF,IAAI;EACJ,IAAI;AACJ,MAAI,QAAQ,QAAQ;AAClB,kBAAe,IAAI,aACjB,QAAQ,OAAO,MACf,QAAQ,OAAO,MAChB;AACD,WAAQ,MAAM,cAAc,cAAc;IACxC,cAAc,QAAQ,OAAO;IAC7B,SAAS,eAAe,IAAI,KAAK,IAAI,CAAC,aAAa;IACnD,WAAW;IACZ,CAAC;AACF,QAAK;IAAE,MAAM;IAAe;IAAO,CAAC;;EAkCtC,MAAM,WAvBU,MAAM,QACpB,QACA,aACA,OAAO,EAAE,GAAG,KAAK,aAAa,UAAU;GACtC,MAAM,cAA4B,EAAE;GACpC,MAAM,cAA+B;IACnC,GAAG;IACH,WAAW,QAAQ,aAAa,QAAQ,IAAI,EAAE,MAAM,EAAE;IACvD;AACD,SAAM,cACJ,GACA,KACA,WACA,OACA,OACA,OACA,aACA,MACA,YACD;AACD,UAAO;IAEV,EACuB,MAAM;EAE9B,MAAM,UAA0B,EAAE,SAAS;EAC3C,MAAM,SAAS,MAAM,eAAe,cAAc,OAAO,SAAS,QAAQ;AAC1E,MAAI,OAAQ,SAAQ,SAAS;AAC7B,SAAO;WACC;AACR,iBAAe"}
@@ -168,12 +168,50 @@ interface EvalDefinition {
168
168
  /** The eval body: drive the agent and assert on its behavior. */
169
169
  test(t: TestContext): Promise<void> | void;
170
170
  }
171
- /** Per-directory config from `evals.config.ts` (see {@link defineEvalConfig}). */
171
+ /**
172
+ * Auto-start config for the app under test, à la Playwright's `webServer`. When
173
+ * set in a root `evals.config.ts`, the CLI boots the app before running evals
174
+ * and tears it down after — so you don't have to start the server by hand.
175
+ */
176
+ interface EvalWebServer {
177
+ /** Shell command that starts the app, e.g. `"npm run dev"`. */
178
+ command: string;
179
+ /**
180
+ * URL polled until it answers before evals start. Defaults to the run's
181
+ * `baseUrl` (`--url`). Readiness = any HTTP response (a 404 still proves the
182
+ * server is up).
183
+ */
184
+ url?: string;
185
+ /** How long to wait for `url` to answer before giving up. Defaults to 60s. */
186
+ timeoutMs?: number;
187
+ /**
188
+ * When `true` (default), reuse a server already answering at `url` instead of
189
+ * spawning one — so a running `dev` server is used as-is. Set `false` to
190
+ * always spawn a fresh server.
191
+ */
192
+ reuseExisting?: boolean;
193
+ }
194
+ /**
195
+ * Eval config from `evals.config.ts` (via {@link defineEvalConfig}).
196
+ *
197
+ * Two scopes share this shape: a **root** `evals.config.ts` (project root) may
198
+ * set run-wide settings — `baseUrl` and `webServer` — plus `maxConcurrency`/
199
+ * `timeoutMs`; a **per-agent** `server/agents/<id>/evals/evals.config.ts` may set
200
+ * that agent's `maxConcurrency`/`timeoutMs` (`baseUrl`/`webServer` there are
201
+ * ignored — server lifecycle is run-wide). Precedence for the shared numeric
202
+ * fields is CLI flag > root config > per-agent config > built-in default, so a
203
+ * per-agent value takes effect only when neither the flag nor the root config
204
+ * sets that field.
205
+ */
172
206
  interface EvalConfig {
173
207
  /** Max evals to run concurrently. */
174
208
  maxConcurrency?: number;
175
209
  /** Default per-eval timeout. */
176
210
  timeoutMs?: number;
211
+ /** Base URL of the app to drive (root config only). Overridden by `--url`. */
212
+ baseUrl?: string;
213
+ /** Auto-start the app under test (root config only). */
214
+ webServer?: EvalWebServer;
177
215
  }
178
216
  /** The outcome of running one eval. */
179
217
  interface EvalResult {
@@ -197,5 +235,5 @@ interface EvalResult {
197
235
  traceId?: string;
198
236
  }
199
237
  //#endregion
200
- export { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalConfig, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext };
238
+ export { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalConfig, EvalDefinition, EvalDriver, EvalResult, EvalWebServer, MatchResult, Matcher, Severity, TestContext };
201
239
  //# sourceMappingURL=types.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;EAQoB;EAN5B,IAAA,IAAQ,eAAA;EAMmC;;;;;EAA3C,OAAA,CAAQ,SAAA,WAAoB,eAAA;AAAA;;UAIb,WAAA;EAJ4B;EAM3C,KAAA;EAF0B;EAI1B,SAAA;EAEsB;EAAtB,eAAA,EAAiB,KAAA;IAAQ,IAAA;IAAc,IAAA,EAAM,MAAA;EAAA;EAApB;EAEzB,SAAA;EAF6C;EAI7C,SAAA;EAAA;EAEA,OAAA;AAAA;;AAOF;;;UAAiB,UAAA;EASJ;;;;;EAHX,IAAA,CACE,OAAA,UACA,OAAA;IAAY,MAAA,GAAS,WAAA;EAAA,IACpB,OAAA,CAAQ,WAAA;EADT;;;;EAMF,KAAA;AAAA;AAIF;AAAA,UAAiB,WAAA;;EAEf,IAAA,CAAK,OAAA,WAAkB,OAAA;EAiBP;;;;;EAXhB,KAAA;EAgC8B;EAAA,SA9BrB,KAAA;EAwC+B;EAAA,SAtC/B,SAAA;EAwC6B;EAAA,SAtC7B,SAAA;EAwCM;;;;EAAA,SAnCN,KAAA,EAAO,MAAA;EAjBhB;;;;EAAA,SAsBS,QAAA,EAAU,MAAA;EAZV;EAcT,SAAA,IAAa,eAAA;EAPJ;EAST,UAAA,CAAW,IAAA,WAAe,eAAA;EAJjB;;;;;;EAWT,cAAA,CACE,IAAA,UACA,QAAA,EAAU,MAAA,oBACT,eAAA;EAHH;EAKA,KAAA,CAAM,KAAA,UAAe,OAAA,EAAS,OAAA,GAAU,eAAA;EAH5B;;;;;;;EAWZ,KAAA;IAAA,mEAEE,UAAA,CAAW,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAA3B;IAEX,QAAA,CAAS,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAFE;IAItC,MAAA,CAAO,IAAA,EAAM,eAAA,GAAkB,OAAA,CAAQ,eAAA;EAAA;EAFX;EAK9B,IAAA,CAAK,MAAA;AAAA;;UAIU,eAAA;EACf,IAAA;EACA,cAAA;EACA,YAAA,EAAc,MAAA;AAAA;;UAIC,cAAA;EAPA;EASf,WAAA;;EAEA,KAAA;EAVA;EAYA,IAAA;EAVA;;;;EAeA,SAAA;EAX6B;;;;;;EAkB7B,OAAA;IAAY,KAAA;IAAe,KAAA;EAAA;EAE3B;EAAA,IAAA,CAAK,CAAA,EAAG,WAAA,GAAc,OAAA;AAAA;;UAIP,UAAA;EAJc;EAM7B,cAAA;EAFyB;EAIzB,SAAA;AAAA;;UAIe,UAAA;EACf,EAAA;EACA,WAAA;EAG2B;EAD3B,OAAA;IAAY,MAAA;EAAA;EACZ,UAAA,EAAY,eAAA;EAAZ;EAEA,MAAA;EAAA;EAEA,KAAA;EAKA;;;;EAAA,YAAA;;EAEA,OAAA;AAAA"}
1
+ {"version":3,"file":"types.d.ts","names":[],"sources":["../../src/evals/types.ts"],"mappings":";;AAWA;;;;;;;;;UAAiB,WAAA;EACf,IAAA;;EAEA,KAAA;EAMkD;EAJlD,MAAA;AAAA;;KAIU,OAAA,IAAW,KAAA,aAAkB,WAAA;;KAG7B,QAAA;;UAGK,eAAA;EACf,KAAA;EACA,QAAA,EAAU,QAAA;EACV,IAAA;EACA,KAAA;EACA,MAAA;AAAA;;;;AAQF;;UAAiB,eAAA;EAEP;EAAR,IAAA,IAAQ,eAAA;EAQoB;EAN5B,IAAA,IAAQ,eAAA;EAMmC;;;;;EAA3C,OAAA,CAAQ,SAAA,WAAoB,eAAA;AAAA;;UAIb,WAAA;EAJ4B;EAM3C,KAAA;EAF0B;EAI1B,SAAA;EAEsB;EAAtB,eAAA,EAAiB,KAAA;IAAQ,IAAA;IAAc,IAAA,EAAM,MAAA;EAAA;EAApB;EAEzB,SAAA;EAF6C;EAI7C,SAAA;EAAA;EAEA,OAAA;AAAA;;AAOF;;;UAAiB,UAAA;EASJ;;;;;EAHX,IAAA,CACE,OAAA,UACA,OAAA;IAAY,MAAA,GAAS,WAAA;EAAA,IACpB,OAAA,CAAQ,WAAA;EADT;;;;EAMF,KAAA;AAAA;AAIF;AAAA,UAAiB,WAAA;;EAEf,IAAA,CAAK,OAAA,WAAkB,OAAA;EAiBP;;;;;EAXhB,KAAA;EAgC8B;EAAA,SA9BrB,KAAA;EAwC+B;EAAA,SAtC/B,SAAA;EAwC6B;EAAA,SAtC7B,SAAA;EAwCM;;;;EAAA,SAnCN,KAAA,EAAO,MAAA;EAjBhB;;;;EAAA,SAsBS,QAAA,EAAU,MAAA;EAZV;EAcT,SAAA,IAAa,eAAA;EAPJ;EAST,UAAA,CAAW,IAAA,WAAe,eAAA;EAJjB;;;;;;EAWT,cAAA,CACE,IAAA,UACA,QAAA,EAAU,MAAA,oBACT,eAAA;EAHH;EAKA,KAAA,CAAM,KAAA,UAAe,OAAA,EAAS,OAAA,GAAU,eAAA;EAH5B;;;;;;;EAWZ,KAAA;IAAA,mEAEE,UAAA,CAAW,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAA3B;IAEX,QAAA,CAAS,QAAA,WAAmB,OAAA,CAAQ,eAAA,GAFE;IAItC,MAAA,CAAO,IAAA,EAAM,eAAA,GAAkB,OAAA,CAAQ,eAAA;EAAA;EAFX;EAK9B,IAAA,CAAK,MAAA;AAAA;;UAIU,eAAA;EACf,IAAA;EACA,cAAA;EACA,YAAA,EAAc,MAAA;AAAA;;UAIC,cAAA;EAPA;EASf,WAAA;;EAEA,KAAA;EAVA;EAYA,IAAA;EAVA;;;;EAeA,SAAA;EAX6B;;;;;;EAkB7B,OAAA;IAAY,KAAA;IAAe,KAAA;EAAA;EAE3B;EAAA,IAAA,CAAK,CAAA,EAAG,WAAA,GAAc,OAAA;AAAA;;;;AAQxB;;UAAiB,aAAA;EAAa;EAE5B,OAAA;EAMA;;;;;EAAA,GAAA;EAuByB;EArBzB,SAAA;EA6ByB;;;;;EAvBzB,aAAA;AAAA;;AA2BF;;;;;;;;;;;UAZiB,UAAA;EA0Bf;EAxBA,cAAA;EA0BO;EAxBP,SAAA;;EAEA,OAAA;;EAEA,SAAA,GAAY,aAAA;AAAA;;UAIG,UAAA;EACf,EAAA;EACA,WAAA;;EAEA,OAAA;IAAY,MAAA;EAAA;EACZ,UAAA,EAAY,eAAA;;EAEZ,MAAA;;EAEA,KAAA;;;;;EAKA,YAAA;;EAEA,OAAA;AAAA"}
package/dist/index.js CHANGED
@@ -1,7 +1,6 @@
1
1
  import { isSQLTypeMarker, sql } from "./shared/src/sql/helpers.js";
2
2
  import { ApiError } from "./shared/src/workspace-client/errors.js";
3
3
  import { createWorkspaceClient } from "./shared/src/workspace-client/factory.js";
4
- import { createLakebasePoolManager } from "./connectors/lakebase/pool-manager.js";
5
4
  import { AppKitError } from "./errors/base.js";
6
5
  import { AuthenticationError } from "./errors/authentication.js";
7
6
  import { ConfigurationError } from "./errors/configuration.js";
@@ -13,6 +12,7 @@ import { ServerError } from "./errors/server.js";
13
12
  import { TunnelError } from "./errors/tunnel.js";
14
13
  import { ValidationError } from "./errors/validation.js";
15
14
  import "./errors/index.js";
15
+ import { createLakebasePoolManager } from "./connectors/lakebase/pool-manager.js";
16
16
  import { getExecutionContext } from "./context/execution-context.js";
17
17
  import { RequestedClaimsPermissionSet, createLakebasePool, generateDatabaseCredential, getLakebaseOrmConfig, getLakebasePgConfig, getUsernameWithApiLookup, getWorkspaceClient } from "./connectors/lakebase/index.js";
18
18
  import { SeverityNumber, SpanStatusCode } from "./telemetry/index.js";
@@ -1,6 +1,6 @@
1
- import { createLogger } from "../logging/logger.js";
2
1
  import { TunnelError } from "../errors/tunnel.js";
3
2
  import "../errors/index.js";
3
+ import { createLogger } from "../logging/logger.js";
4
4
  import { isRemoteTunnelAllowedByEnv } from "../plugins/server/remote-tunnel/gate.js";
5
5
  import { randomUUID } from "node:crypto";
6
6
 
@@ -1,5 +1,5 @@
1
- import { createLogger } from "../../logging/logger.js";
2
1
  import { AppKitError } from "../../errors/base.js";
2
+ import { createLogger } from "../../logging/logger.js";
3
3
 
4
4
  //#region src/plugin/interceptors/retry.ts
5
5
  const logger = createLogger("interceptors:retry");
@@ -1,9 +1,9 @@
1
1
  import { camelToKebab } from "../shared/src/naming.js";
2
- import { createLogger } from "../logging/logger.js";
3
2
  import { AppKitError } from "../errors/base.js";
4
3
  import { AuthenticationError } from "../errors/authentication.js";
5
4
  import "../errors/index.js";
6
5
  import { ServiceContext } from "../context/service-context.js";
6
+ import { createLogger } from "../logging/logger.js";
7
7
  import { getCurrentUserId, runInUserContext } from "../context/execution-context.js";
8
8
  import { normalizeTelemetryOptions } from "../telemetry/config.js";
9
9
  import { TelemetryManager } from "../telemetry/telemetry-manager.js";
@@ -1,8 +1,8 @@
1
1
  import { makeResultMessage } from "../../shared/src/sse/analytics.js";
2
- import { createLogger } from "../../logging/logger.js";
3
2
  import { AppKitError } from "../../errors/base.js";
4
3
  import { ExecutionError } from "../../errors/execution.js";
5
4
  import "../../errors/index.js";
5
+ import { createLogger } from "../../logging/logger.js";
6
6
  import { getWarehouseId, getWorkspaceClient } from "../../context/execution-context.js";
7
7
  import "../../context/index.js";
8
8
  import { Plugin } from "../../plugin/plugin.js";
@@ -1,6 +1,6 @@
1
- import { createLogger } from "../../logging/logger.js";
2
1
  import { ExecutionError } from "../../errors/execution.js";
3
2
  import "../../errors/index.js";
3
+ import { createLogger } from "../../logging/logger.js";
4
4
  import { Type, tableFromIPC } from "apache-arrow";
5
5
 
6
6
  //#region src/plugins/analytics/result-delivery.ts
@@ -1,6 +1,4 @@
1
1
  import { agents } from "./agents/agents.js";
2
2
  import "./agents/index.js";
3
3
  import { aiSearch } from "./ai-search/ai-search.js";
4
- import "./ai-search/index.js";
5
- import { database } from "./database/database.js";
6
- import "./database/index.js";
4
+ import "./ai-search/index.js";
@@ -2,7 +2,5 @@ import { agents } from "./agents/agents.js";
2
2
  import "./agents/index.js";
3
3
  import { aiSearch } from "./ai-search/ai-search.js";
4
4
  import "./ai-search/index.js";
5
- import { database } from "./database/database.js";
6
- import "./database/index.js";
7
5
 
8
6
  export { };
@@ -0,0 +1,14 @@
1
+ import { databaseSetupFailed } from "../../database/errors.js";
2
+
3
+ //#region src/plugins/database/config.ts
4
+ /** Reject invalid arguments instead of turning them into the default API. */
5
+ function assertDatabaseConfig(config) {
6
+ if (typeof config !== "object" || config === null || Array.isArray(config)) throw databaseSetupFailed("Expected a database configuration object or no argument.");
7
+ const prototype = Object.getPrototypeOf(config);
8
+ if (prototype !== Object.prototype && prototype !== null) throw databaseSetupFailed("Expected a plain database configuration object.");
9
+ if ("crudRoutes" in config) throw databaseSetupFailed("\"crudRoutes\" was renamed to \"api\". Use api: false to disable generated routes or api: { writes: false } for reads only.");
10
+ }
11
+
12
+ //#endregion
13
+ export { assertDatabaseConfig };
14
+ //# sourceMappingURL=config.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"config.js","names":[],"sources":["../../../src/plugins/database/config.ts"],"sourcesContent":["import { databaseSetupFailed } from \"../../database/errors\";\n\n/** Reject invalid arguments instead of turning them into the default API. */\nexport function assertDatabaseConfig(config: unknown): void {\n if (typeof config !== \"object\" || config === null || Array.isArray(config)) {\n throw databaseSetupFailed(\n \"Expected a database configuration object or no argument.\",\n );\n }\n const prototype = Object.getPrototypeOf(config);\n if (prototype !== Object.prototype && prototype !== null) {\n throw databaseSetupFailed(\n \"Expected a plain database configuration object.\",\n );\n }\n // Do not silently turn a previous opt-out into the default full API.\n if (\"crudRoutes\" in config) {\n throw databaseSetupFailed(\n '\"crudRoutes\" was renamed to \"api\". Use api: false to disable generated routes or api: { writes: false } for reads only.',\n );\n }\n}\n"],"mappings":";;;;AAGA,SAAgB,qBAAqB,QAAuB;AAC1D,KAAI,OAAO,WAAW,YAAY,WAAW,QAAQ,MAAM,QAAQ,OAAO,CACxE,OAAM,oBACJ,2DACD;CAEH,MAAM,YAAY,OAAO,eAAe,OAAO;AAC/C,KAAI,cAAc,OAAO,aAAa,cAAc,KAClD,OAAM,oBACJ,kDACD;AAGH,KAAI,gBAAgB,OAClB,OAAM,oBACJ,8HACD"}
@@ -6,12 +6,12 @@ import "../../plugin/index.js";
6
6
  import { PluginManifest } from "../../registry/types.js";
7
7
  import "../../registry/index.js";
8
8
  import { DatabaseExports } from "./entity-types.js";
9
- import { IDatabaseConfig } from "./types.js";
9
+ import { DefaultDatabaseSchema, IDatabaseConfig } from "./types.js";
10
10
  import express from "express";
11
11
 
12
12
  //#region src/plugins/database/database.d.ts
13
13
  /** Schema-driven database plugin */
14
- declare class DatabasePlugin<TSchema extends Schema> extends Plugin<IDatabaseConfig<TSchema>> {
14
+ declare class DatabasePlugin<TSchema extends Schema = DefaultDatabaseSchema> extends Plugin<IDatabaseConfig<TSchema>> {
15
15
  /** Plugin metadata and required PostgreSQL resource. */
16
16
  static manifest: PluginManifest<"database">;
17
17
  protected config: IDatabaseConfig<TSchema>;
@@ -20,7 +20,8 @@ declare class DatabasePlugin<TSchema extends Schema> extends Plugin<IDatabaseCon
20
20
  private draining;
21
21
  private shutdownPromise;
22
22
  private exposure;
23
- constructor(config: IDatabaseConfig<TSchema>);
23
+ private resolvedSchema;
24
+ constructor(config?: IDatabaseConfig<TSchema>);
24
25
  /** Build and verify one candidate state before publishing its exports. */
25
26
  setup(): Promise<void>;
26
27
  /** Register generated CRUD, subject to the configured table and write restrictions. */
@@ -34,12 +35,28 @@ declare class DatabasePlugin<TSchema extends Schema> extends Plugin<IDatabaseCon
34
35
  /** Trace one generated route with allowlisted, low-cardinality attributes. */
35
36
  private runRouteSpan;
36
37
  }
37
- /** Create a typed database plugin registration for a finalized schema. */
38
- declare function database<TSchema extends Schema>(config: IDatabaseConfig<TSchema>): {
38
+ type DatabaseRegistration<TSchema extends Schema> = {
39
39
  plugin: PluginConstructor<BasePluginConfig, DatabasePlugin<TSchema>>;
40
40
  config: IDatabaseConfig<TSchema>;
41
41
  name: "database";
42
42
  };
43
+ /**
44
+ * Create the database plugin. Omit configuration to load
45
+ * `config/database/schema.ts`, or supply a typed schema override.
46
+ */
47
+ declare function database<TSchema extends Schema>(config: IDatabaseConfig<TSchema> & {
48
+ readonly schema: TSchema;
49
+ }): DatabaseRegistration<TSchema> & {
50
+ config: IDatabaseConfig<TSchema> & {
51
+ readonly schema: TSchema;
52
+ };
53
+ };
54
+ /**
55
+ * Register the database plugin with opinionated defaults and full HTTP CRUD.
56
+ * By default, setup loads the named `schema` export from the application's
57
+ * `config/database/schema.ts`. Registration itself performs no file or database I/O.
58
+ */
59
+ declare function database<TSchema extends Schema = DefaultDatabaseSchema>(config?: IDatabaseConfig<TSchema>): DatabaseRegistration<TSchema>;
43
60
  //#endregion
44
61
  export { database };
45
62
  //# sourceMappingURL=database.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"database.d.ts","names":[],"sources":["../../../src/plugins/database/database.ts"],"mappings":";;;;;;;;;;;;;cA6Ba,cAAA,iBAA+B,MAAA,UAAgB,MAAA,CAC1D,eAAA,CAAgB,OAAA;;SAGT,QAAA,EAAuB,cAAA;EAAA,UACZ,MAAA,EAAQ,eAAA,CAAgB,OAAA;EAAA,QAClC,KAAA;EAAA,QACA,YAAA;EAAA,QACA,QAAA;EAAA,QACA,eAAA;EAAA,QACA,QAAA;cAEI,MAAA,EAAQ,eAAA,CAAgB,OAAA;EARN;EAwBxB,KAAA,CAAA,GAAS,OAAA;EAvBW;EA0D1B,YAAA,CAAa,MAAA,EAAQ,OAAA,CAAQ,MAAA;EAnDT;EAAA,QAqHZ,KAAA;EAlEa;EAuErB,OAAA,CAAA,GAOO,eAAA;EAIW;EAAZ,QAAA,CAAA,GAAY,OAAA;EAjJ8C;EAAA,QAqKxD,YAAA;AAAA;;iBA+BM,QAAA,iBAAyB,MAAA,CAAA,CACvC,MAAA,EAAQ,eAAA,CAAgB,OAAA;UAGe,iBAAA,CACnC,gBAAA,EACA,cAAA,CAAe,OAAA"}
1
+ {"version":3,"file":"database.d.ts","names":[],"sources":["../../../src/plugins/database/database.ts"],"mappings":";;;;;;;;;;;;;cAoCa,cAAA,iBACK,MAAA,GAAS,qBAAA,UACjB,MAAA,CAAO,eAAA,CAAgB,OAAA;;SAExB,QAAA,EAAuB,cAAA;EAAA,UACZ,MAAA,EAAQ,eAAA,CAAgB,OAAA;EAAA,QAClC,KAAA;EAAA,QACA,YAAA;EAAA,QACA,QAAA;EAAA,QACA,eAAA;EAAA,QACA,QAAA;EAAA,QACA,cAAA;cAEI,MAAA,GAAQ,eAAA,CAAgB,OAAA;EATN;EAoBxB,KAAA,CAAA,GAAS,OAAA;EAnBW;EAoE1B,YAAA,CAAa,MAAA,EAAQ,OAAA,CAAQ,MAAA;EA5DT;EAAA,QA6HZ,KAAA;EAjEa;EAsErB,OAAA,CAAA,GAOO,eAAA;EAIW;EAAZ,QAAA,CAAA,GAAY,OAAA;EAxJJ;EAAA,QA4KN,YAAA;AAAA;AAAA,KA8BL,oBAAA,iBAAqC,MAAA;EACxC,MAAA,EAAQ,iBAAA,CAAkB,gBAAA,EAAkB,cAAA,CAAe,OAAA;EAC3D,MAAA,EAAQ,eAAA,CAAgB,OAAA;EACxB,IAAA;AAAA;;;;;iBAOc,QAAA,iBAAyB,MAAA,CAAA,CACvC,MAAA,EAAQ,eAAA,CAAgB,OAAA;EAAA,SAAsB,MAAA,EAAQ,OAAA;AAAA,IACrD,oBAAA,CAAqB,OAAA;EACtB,MAAA,EAAQ,eAAA,CAAgB,OAAA;IAAA,SAAsB,MAAA,EAAQ,OAAA;EAAA;AAAA;;;;;;iBAOxC,QAAA,iBAAyB,MAAA,GAAS,qBAAA,CAAA,CAChD,MAAA,GAAS,eAAA,CAAgB,OAAA,IACxB,oBAAA,CAAqB,OAAA"}