@databricks/appkit 0.72.0 → 0.73.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +17 -0
- package/NOTICE.md +1 -0
- package/dist/appkit/package.js +1 -1
- package/dist/beta.d.ts +7 -4
- package/dist/beta.js +3 -2
- package/dist/cli/commands/agent/eval.js +9 -1
- package/dist/cli/commands/agent/eval.js.map +1 -1
- package/dist/connectors/index.js +1 -1
- package/dist/connectors/mlflow/auth.d.ts +11 -1
- package/dist/connectors/mlflow/auth.d.ts.map +1 -1
- package/dist/connectors/mlflow/auth.js +22 -2
- package/dist/connectors/mlflow/auth.js.map +1 -1
- package/dist/connectors/mlflow/index.d.ts +2 -0
- package/dist/database/errors.js +15 -5
- package/dist/database/errors.js.map +1 -1
- package/dist/database/runtime/data-path.d.ts +7 -0
- package/dist/database/runtime/data-path.d.ts.map +1 -0
- package/dist/database/runtime/data-path.js.map +1 -1
- package/dist/database/runtime/engine/drizzle-data-path.js +7 -5
- package/dist/database/runtime/engine/drizzle-data-path.js.map +1 -1
- package/dist/database/schema-builder/define-schema.d.ts +1 -1
- package/dist/database/schema-builder/define-schema.js +1 -1
- package/dist/database/schema-builder/define-schema.js.map +1 -1
- package/dist/errors/database-validation.d.ts +23 -0
- package/dist/errors/database-validation.d.ts.map +1 -0
- package/dist/errors/database-validation.js +24 -0
- package/dist/errors/database-validation.js.map +1 -0
- package/dist/errors/index.js +1 -0
- package/dist/evals/dataset.d.ts +36 -0
- package/dist/evals/dataset.d.ts.map +1 -0
- package/dist/evals/dataset.js +36 -0
- package/dist/evals/dataset.js.map +1 -0
- package/dist/evals/http-driver.js +56 -51
- package/dist/evals/http-driver.js.map +1 -1
- package/dist/evals/index.d.ts +14 -0
- package/dist/evals/index.js +2 -1
- package/dist/evals/judge.d.ts +1 -0
- package/dist/evals/judge.d.ts.map +1 -1
- package/dist/evals/mlflow-report.d.ts +1 -0
- package/dist/evals/mlflow-report.d.ts.map +1 -1
- package/dist/evals/mlflow-run.d.ts +2 -0
- package/dist/evals/mlflow-run.d.ts.map +1 -1
- package/dist/evals/run-eval.d.ts +3 -0
- package/dist/evals/run-eval.d.ts.map +1 -1
- package/dist/evals/run-eval.js +10 -2
- package/dist/evals/run-eval.js.map +1 -1
- package/dist/evals/run-evals.d.ts +9 -0
- package/dist/evals/run-evals.d.ts.map +1 -1
- package/dist/evals/run-evals.js +101 -22
- package/dist/evals/run-evals.js.map +1 -1
- package/dist/evals/types.d.ts +40 -4
- package/dist/evals/types.d.ts.map +1 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/plugin/plugin.d.ts.map +1 -1
- package/dist/plugin/plugin.js +1 -1
- package/dist/plugin/plugin.js.map +1 -1
- package/dist/plugins/database/crud/contract.js +17 -8
- package/dist/plugins/database/crud/contract.js.map +1 -1
- package/dist/plugins/database/crud/exposure.js +63 -22
- package/dist/plugins/database/crud/exposure.js.map +1 -1
- package/dist/plugins/database/crud/request.js +50 -0
- package/dist/plugins/database/crud/request.js.map +1 -0
- package/dist/plugins/database/crud/response.js +77 -0
- package/dist/plugins/database/crud/response.js.map +1 -0
- package/dist/plugins/database/crud/routes.js +71 -52
- package/dist/plugins/database/crud/routes.js.map +1 -1
- package/dist/plugins/database/database.d.ts +6 -4
- package/dist/plugins/database/database.d.ts.map +1 -1
- package/dist/plugins/database/database.js +46 -16
- package/dist/plugins/database/database.js.map +1 -1
- package/dist/plugins/database/defaults.js +5 -1
- package/dist/plugins/database/defaults.js.map +1 -1
- package/dist/plugins/database/entity-client.js +143 -10
- package/dist/plugins/database/entity-client.js.map +1 -1
- package/dist/plugins/database/entity-types.d.ts +1 -1
- package/dist/plugins/database/hooks.d.ts +38 -0
- package/dist/plugins/database/hooks.d.ts.map +1 -0
- package/dist/plugins/database/index.d.ts +3 -2
- package/dist/plugins/database/lifecycle.js +67 -28
- package/dist/plugins/database/lifecycle.js.map +1 -1
- package/dist/plugins/database/scope.js +58 -0
- package/dist/plugins/database/scope.js.map +1 -0
- package/dist/plugins/database/types.d.ts +40 -12
- package/dist/plugins/database/types.d.ts.map +1 -1
- package/dist/shared/src/schemas/manifest.d.ts +33 -33
- package/docs/api/appkit/Class.AppKitError.md +1 -0
- package/docs/api/appkit/Class.DatabaseValidationError.md +191 -0
- package/docs/api/appkit/Function.defineSchema.md +1 -1
- package/docs/api/appkit/Function.readEvalDataset.md +21 -0
- package/docs/api/appkit/Function.resolveWorkspaceClient.md +18 -0
- package/docs/api/appkit/Interface.AssertionHandle.md +1 -1
- package/docs/api/appkit/Interface.DatabaseValidationIssue.md +21 -0
- package/docs/api/appkit/Interface.DatasetRow.md +21 -0
- package/docs/api/appkit/Interface.EntityMutationHooks.md +173 -0
- package/docs/api/appkit/Interface.EvalDefinition.md +28 -0
- package/docs/api/appkit/Interface.EvalDriver.md +16 -1
- package/docs/api/appkit/Interface.HookApp.md +12 -0
- package/docs/api/appkit/Interface.HookContext.md +21 -0
- package/docs/api/appkit/Interface.ReadEvalDatasetOptions.md +34 -0
- package/docs/api/appkit/Interface.ReadSerializerContext.md +21 -0
- package/docs/api/appkit/Interface.RunEvalOptions.md +11 -0
- package/docs/api/appkit/Interface.RunEvalsOptions.md +22 -0
- package/docs/api/appkit/Interface.TestContext.md +41 -4
- package/docs/api/appkit/TypeAlias.DatabaseApiConfig.md +53 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md +8 -0
- package/docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md +49 -0
- package/docs/api/appkit/TypeAlias.DatabaseExports.md +3 -3
- package/docs/api/appkit/TypeAlias.EntityHooks.md +25 -0
- package/docs/api/appkit/TypeAlias.IDatabaseConfig.md +16 -5
- package/docs/api/appkit/TypeAlias.ReadSerializer.md +19 -0
- package/docs/api/appkit/TypeAlias.TransactionClient.md +19 -0
- package/docs/api/appkit.md +135 -119
- package/docs/plugins/database.md +144 -0
- package/llms.txt +17 -0
- package/package.json +2 -2
- package/sbom.cdx.json +1 -1
package/CLAUDE.md
CHANGED
|
@@ -48,6 +48,7 @@ npx @databricks/appkit docs <query>
|
|
|
48
48
|
- [Analytics plugin](./docs/plugins/analytics.md): Enables SQL query execution against Databricks SQL Warehouses.
|
|
49
49
|
- [Caching](./docs/plugins/caching.md): AppKit provides both global and plugin-level caching capabilities.
|
|
50
50
|
- [Creating custom plugins](./docs/plugins/custom-plugins.md): If you need custom API routes or background logic, implement an AppKit plugin. The fastest way is to use the CLI:
|
|
51
|
+
- [Database plugin](./docs/plugins/database.md): This plugin is currently beta. APIs may change between minor releases. Import from @databricks/appkit/beta. See Plugin Stability Tiers.
|
|
51
52
|
- [Execution context](./docs/plugins/execution-context.md): AppKit manages Databricks authentication via two contexts:
|
|
52
53
|
- [Files plugin](./docs/plugins/files.md): File operations against Databricks Unity Catalog Volumes. Supports listing, reading, downloading, uploading, deleting, and previewing files with built-in caching, retry, and timeout handling via the execution interceptor pipeline.
|
|
53
54
|
- [Genie plugin](./docs/plugins/genie.md): Integrates Databricks AI/BI Genie spaces into your AppKit application, enabling natural language data queries via a conversational interface.
|
|
@@ -68,6 +69,7 @@ npx @databricks/appkit docs <query>
|
|
|
68
69
|
- [Class: AuthenticationError](./docs/api/appkit/Class.AuthenticationError.md): Error thrown when authentication fails.
|
|
69
70
|
- [Class: ConfigurationError](./docs/api/appkit/Class.ConfigurationError.md): Error thrown when configuration is missing or invalid.
|
|
70
71
|
- [Class: ConnectionError](./docs/api/appkit/Class.ConnectionError.md): Error thrown when a connection or network operation fails.
|
|
72
|
+
- [Class: DatabaseValidationError](./docs/api/appkit/Class.DatabaseValidationError.md): Deliberate validation failure raised by a database mutation hook. Generated
|
|
71
73
|
- [Class: DatabricksAdapter](./docs/api/appkit/Class.DatabricksAdapter.md): Adapter that talks directly to Databricks Model Serving /invocations endpoint.
|
|
72
74
|
- [Class: ExecutionError](./docs/api/appkit/Class.ExecutionError.md): Error thrown when an operation execution fails.
|
|
73
75
|
- [Class: InitializationError](./docs/api/appkit/Class.InitializationError.md): Error thrown when a service or component is not properly initialized.
|
|
@@ -138,9 +140,11 @@ npx @databricks/appkit docs <query>
|
|
|
138
140
|
- [Function: mcpServer()](./docs/api/appkit/Function.mcpServer.md): Factory for declaring a custom MCP server tool.
|
|
139
141
|
- [Function: normalizeHost()](./docs/api/appkit/Function.normalizeHost.md): Ensure the host has a scheme (Databricks env often lacks https://).
|
|
140
142
|
- [Function: parseTextToolCalls()](./docs/api/appkit/Function.parseTextToolCalls.md): Parses text-based tool calls from model output.
|
|
143
|
+
- [Function: readEvalDataset()](./docs/api/appkit/Function.readEvalDataset.md): Read a Databricks managed evaluation dataset (a Unity Catalog table with
|
|
141
144
|
- [Function: reportToMlflow()](./docs/api/appkit/Function.reportToMlflow.md): Write one pass/fail assessment per eval result to the Databricks MLflow REST
|
|
142
145
|
- [Function: resolveDatabricksAuth()](./docs/api/appkit/Function.resolveDatabricksAuth.md): Parameters
|
|
143
146
|
- [Function: resolveHostedTools()](./docs/api/appkit/Function.resolveHostedTools.md): Parameters
|
|
147
|
+
- [Function: resolveWorkspaceClient()](./docs/api/appkit/Function.resolveWorkspaceClient.md): Construct a Databricks WorkspaceClient for the eval runner — the object the
|
|
144
148
|
- [Function: runAgent()](./docs/api/appkit/Function.runAgent.md): Standalone agent execution without createApp. Resolves the adapter, binds
|
|
145
149
|
- [Function: runEval()](./docs/api/appkit/Function.runEval.md): Run a single eval against a driver. Never throws for assertion or agent
|
|
146
150
|
- [Function: runEvalsInDir()](./docs/api/appkit/Function.runEvalsInDir.md): Discover, load, and run every eval under each agent's evals/ dir, driving
|
|
@@ -166,10 +170,13 @@ npx @databricks/appkit docs <query>
|
|
|
166
170
|
- [Interface: CustomJudgeSpec](./docs/api/appkit/Interface.CustomJudgeSpec.md): A custom LLM-judge definition: a prompt template and choice→score mapping.
|
|
167
171
|
- [Interface: DatabaseCredential](./docs/api/appkit/Interface.DatabaseCredential.md): Database credentials with OAuth token for Postgres connection
|
|
168
172
|
- [Interface: DatabaseRegistry](./docs/api/appkit/Interface.DatabaseRegistry.md): CANONICAL augmentation target. Empty by default; the generated database.d.ts
|
|
173
|
+
- [Interface: DatabaseValidationIssue](./docs/api/appkit/Interface.DatabaseValidationIssue.md): One rejected field; path names public columns, never their values.
|
|
169
174
|
- [Interface: DatabricksAuth](./docs/api/appkit/Interface.DatabricksAuth.md): Resolved Databricks host + bearer token for the eval runner's REST calls.
|
|
175
|
+
- [Interface: DatasetRow](./docs/api/appkit/Interface.DatasetRow.md): One row of a managed evaluation dataset. inputs are the kwargs passed to the
|
|
170
176
|
- [Interface: DiscoveredEval](./docs/api/appkit/Interface.DiscoveredEval.md): An eval file found under server/agents//evals/.
|
|
171
177
|
- [Interface: DriveResult](./docs/api/appkit/Interface.DriveResult.md): What a driver returns for a single t.send.
|
|
172
178
|
- [Interface: EndpointConfig](./docs/api/appkit/Interface.EndpointConfig.md): Properties
|
|
179
|
+
- [Interface: EntityMutationHooks<TTable>](./docs/api/appkit/Interface.EntityMutationHooks.md): Mutation lifecycle for one entity. A before hook may return a replacement
|
|
173
180
|
- [Interface: EvalDefinition](./docs/api/appkit/Interface.EvalDefinition.md): A single eval, default-exported from a *.eval.ts file.
|
|
174
181
|
- [Interface: EvalDriver](./docs/api/appkit/Interface.EvalDriver.md): Abstraction over how the agent is driven. The HTTP driver posts to a running
|
|
175
182
|
- [Interface: EvalResult](./docs/api/appkit/Interface.EvalResult.md): The outcome of running one eval.
|
|
@@ -180,6 +187,8 @@ npx @databricks/appkit docs <query>
|
|
|
180
187
|
- [Interface: FunctionTool](./docs/api/appkit/Interface.FunctionTool.md): Properties
|
|
181
188
|
- [Interface: GenerateDatabaseCredentialRequest](./docs/api/appkit/Interface.GenerateDatabaseCredentialRequest.md): Request parameters for generating database OAuth credentials
|
|
182
189
|
- [Interface: GenerationParams](./docs/api/appkit/Interface.GenerationParams.md): Optional generation parameters forwarded to the OpenAI-compatible serving
|
|
190
|
+
- [Interface: HookApp](./docs/api/appkit/Interface.HookApp.md): The only capability a hook receives: entities bound to its transaction.
|
|
191
|
+
- [Interface: HookContext](./docs/api/appkit/Interface.HookContext.md): Which entity is being mutated, and the surface a hook may write through.
|
|
183
192
|
- [Interface: HostedSupervisorTool](./docs/api/appkit/Interface.HostedSupervisorTool.md): Tagged record returned by every supervisorTools factory. The
|
|
184
193
|
- [Interface: HttpDriverOptions](./docs/api/appkit/Interface.HttpDriverOptions.md): Properties
|
|
185
194
|
- [Interface: IAiSearchConfig](./docs/api/appkit/Interface.IAiSearchConfig.md): Base configuration interface for AppKit plugins
|
|
@@ -201,6 +210,8 @@ npx @databricks/appkit docs <query>
|
|
|
201
210
|
- [Interface: PluginToolkitProvider](./docs/api/appkit/Interface.PluginToolkitProvider.md): Minimum shape every entry in the Plugins map must expose. Core
|
|
202
211
|
- [Interface: PostResult](./docs/api/appkit/Interface.PostResult.md): Structured result for a best-effort POST that must not throw.
|
|
203
212
|
- [Interface: PromptContext](./docs/api/appkit/Interface.PromptContext.md): Context passed to baseSystemPrompt callbacks.
|
|
213
|
+
- [Interface: ReadEvalDatasetOptions](./docs/api/appkit/Interface.ReadEvalDatasetOptions.md): Properties
|
|
214
|
+
- [Interface: ReadSerializerContext](./docs/api/appkit/Interface.ReadSerializerContext.md): Which entity and generated operation produced the row being shaped.
|
|
204
215
|
- [Interface: RegisteredAgent](./docs/api/appkit/Interface.RegisteredAgent.md): Properties
|
|
205
216
|
- [Interface: ReportOutcome](./docs/api/appkit/Interface.ReportOutcome.md): Properties
|
|
206
217
|
- [Interface: RequestedClaims](./docs/api/appkit/Interface.RequestedClaims.md): Optional claims for fine-grained Unity Catalog table permissions
|
|
@@ -242,7 +253,11 @@ npx @databricks/appkit docs <query>
|
|
|
242
253
|
- [Type Alias: AgentToolsFn()](./docs/api/appkit/TypeAlias.AgentToolsFn.md): Function form of AgentDefinition.tools. Receives the typed
|
|
243
254
|
- [Type Alias: BaseSystemPromptOption](./docs/api/appkit/TypeAlias.BaseSystemPromptOption.md)
|
|
244
255
|
- [Type Alias: ConfigSchema](./docs/api/appkit/TypeAlias.ConfigSchema.md): Configuration schema definition for plugin config.
|
|
256
|
+
- [Type Alias: DatabaseApiConfig<TSchema>](./docs/api/appkit/TypeAlias.DatabaseApiConfig.md): Full generated CRUD for every declared table by default. Set false to disable
|
|
257
|
+
- [Type Alias: DatabaseApiWriteOperation](./docs/api/appkit/TypeAlias.DatabaseApiWriteOperation.md): Generated HTTP write operations.
|
|
258
|
+
- [Type Alias: DatabaseApiWritesConfig<TSchema>](./docs/api/appkit/TypeAlias.DatabaseApiWritesConfig.md): All writes by default; false keeps reads only, and an object narrows writes.
|
|
245
259
|
- [Type Alias: DatabaseExports](./docs/api/appkit/TypeAlias.DatabaseExports.md): Typed database API published by the plugin.
|
|
260
|
+
- [Type Alias: EntityHooks<TTable>](./docs/api/appkit/TypeAlias.EntityHooks.md): Response shaping and mutation lifecycle declared for one table.
|
|
246
261
|
- [Type Alias: EvalProgress](./docs/api/appkit/TypeAlias.EvalProgress.md)
|
|
247
262
|
- [Type Alias: ExecutionResult<T>](./docs/api/appkit/TypeAlias.ExecutionResult.md): Discriminated union for plugin execution results.
|
|
248
263
|
- [Type Alias: FileAction](./docs/api/appkit/TypeAlias.FileAction.md): Every action the files plugin can perform.
|
|
@@ -254,6 +269,7 @@ npx @databricks/appkit docs <query>
|
|
|
254
269
|
- [Type Alias: Matcher()](./docs/api/appkit/TypeAlias.Matcher.md): A deterministic matcher: inspects a string value and returns a result.
|
|
255
270
|
- [Type Alias: PluginData<T, U, N>](./docs/api/appkit/TypeAlias.PluginData.md): Tuple of plugin class, config, and name. Created by toPlugin() and passed to createApp().
|
|
256
271
|
- [Type Alias: Plugins](./docs/api/appkit/TypeAlias.Plugins.md): Plugin map passed to the function form of AgentDefinition.tools.
|
|
272
|
+
- [Type Alias: ReadSerializer()](./docs/api/appkit/TypeAlias.ReadSerializer.md): Shape one already private-safe row before it reaches the wire. A Promise
|
|
257
273
|
- [Type Alias: ResolvedToolEntry](./docs/api/appkit/TypeAlias.ResolvedToolEntry.md): Internal tool-index entry after a tool record has been resolved to a dispatchable form.
|
|
258
274
|
- [Type Alias: ResourceFieldEntry](./docs/api/appkit/TypeAlias.ResourceFieldEntry.md)
|
|
259
275
|
- [Type Alias: ResourcePermission](./docs/api/appkit/TypeAlias.ResourcePermission.md): Union of all possible permission levels across all resource types.
|
|
@@ -263,6 +279,7 @@ npx @databricks/appkit docs <query>
|
|
|
263
279
|
- [Type Alias: SupervisorTool](./docs/api/appkit/TypeAlias.SupervisorTool.md): Tools supported by the Databricks AI Gateway Responses API. The shapes match
|
|
264
280
|
- [Type Alias: ToolRegistry](./docs/api/appkit/TypeAlias.ToolRegistry.md)
|
|
265
281
|
- [Type Alias: ToPlugin()<T, U, N>](./docs/api/appkit/TypeAlias.ToPlugin.md): Factory function type returned by toPlugin(). Accepts optional config and returns a PluginData tuple.
|
|
282
|
+
- [Type Alias: TransactionClient](./docs/api/appkit/TypeAlias.TransactionClient.md): Entity and SQL capabilities bound to one transaction.
|
|
266
283
|
- [Variable: agents](./docs/api/appkit/Variable.agents.md): Plugin factory for the agents plugin. Discovers agents from
|
|
267
284
|
- [Variable: aiSearch](./docs/api/appkit/Variable.aiSearch.md)
|
|
268
285
|
- [Variable: READ_ACTIONS](./docs/api/appkit/Variable.READ_ACTIONS.md): Actions that only read data.
|
package/NOTICE.md
CHANGED
|
@@ -54,6 +54,7 @@ This Software contains code from the following open source projects:
|
|
|
54
54
|
| [@tanstack/react-table](https://www.npmjs.com/package/@tanstack/react-table) | 8.21.3 | MIT | https://tanstack.com/table |
|
|
55
55
|
| [@types/semver](https://www.npmjs.com/package/@types/semver) | 7.7.1 | MIT | https://github.com/DefinitelyTyped/DefinitelyTyped/tree/master/types/semver |
|
|
56
56
|
| [apache-arrow](https://www.npmjs.com/package/apache-arrow) | 21.1.0 | Apache-2.0 | https://arrow.apache.org/js/ |
|
|
57
|
+
| [autoevals](https://www.npmjs.com/package/autoevals) | 0.3.0 | MIT | https://www.braintrust.dev/docs |
|
|
57
58
|
| [class-variance-authority](https://www.npmjs.com/package/class-variance-authority) | 0.7.1 | Apache-2.0 | https://github.com/joe-bell/cva#readme |
|
|
58
59
|
| [clsx](https://www.npmjs.com/package/clsx) | 2.1.1 | MIT | https://github.com/lukeed/clsx#readme |
|
|
59
60
|
| [cmdk](https://www.npmjs.com/package/cmdk) | 1.1.1 | MIT | https://github.com/pacocoursey/cmdk#readme |
|
package/dist/appkit/package.js
CHANGED
package/dist/beta.d.ts
CHANGED
|
@@ -15,8 +15,9 @@ import { Schema } from "./database/schema-builder/types.js";
|
|
|
15
15
|
import { bigid, bigint, boolean, enumColumn, id, integer, jsonb, text, timestamp, uuid, varchar } from "./database/schema-builder/columns.js";
|
|
16
16
|
import { defineSchema } from "./database/schema-builder/define-schema.js";
|
|
17
17
|
import { fk } from "./database/schema-builder/fk.js";
|
|
18
|
-
import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth } from "./connectors/mlflow/auth.js";
|
|
18
|
+
import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth, resolveWorkspaceClient } from "./connectors/mlflow/auth.js";
|
|
19
19
|
import { MlflowClient, PostResult, normalizeHost } from "./connectors/mlflow/client.js";
|
|
20
|
+
import { DatasetRow, ReadEvalDatasetOptions, readEvalDataset } from "./evals/dataset.js";
|
|
20
21
|
import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext } from "./evals/types.js";
|
|
21
22
|
import { defineEval } from "./evals/define-eval.js";
|
|
22
23
|
import { DiscoveredEval, discoverEvalFiles } from "./evals/discover.js";
|
|
@@ -27,14 +28,16 @@ import { Assessment, ReportOutcome, buildAssessments, reportToMlflow } from "./e
|
|
|
27
28
|
import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize } from "./evals/report.js";
|
|
28
29
|
import { RunEvalOptions, runEval } from "./evals/run-eval.js";
|
|
29
30
|
import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir } from "./evals/run-evals.js";
|
|
31
|
+
import "./evals/index.js";
|
|
30
32
|
import { agentIdFromMarkdownPath, loadAgentFromFile, loadAgentsFromDir } from "./core/agent/load-agents.js";
|
|
31
33
|
import { agents } from "./plugins/agents/agents.js";
|
|
32
34
|
import "./plugins/agents/index.js";
|
|
33
35
|
import { IAiSearchConfig, IndexConfig, RerankerConfig, SearchFilters, SearchRequest, SearchResponse, SearchResult } from "./plugins/ai-search/types.js";
|
|
34
36
|
import { aiSearch } from "./plugins/ai-search/ai-search.js";
|
|
35
|
-
import { DatabaseExports } from "./plugins/database/entity-types.js";
|
|
36
|
-
import {
|
|
37
|
+
import { DatabaseExports, TransactionClient } from "./plugins/database/entity-types.js";
|
|
38
|
+
import { EntityMutationHooks, HookApp, HookContext } from "./plugins/database/hooks.js";
|
|
39
|
+
import { DatabaseApiConfig, DatabaseApiWriteOperation, DatabaseApiWritesConfig, EntityHooks, IDatabaseConfig, ReadSerializer, ReadSerializerContext } from "./plugins/database/types.js";
|
|
37
40
|
import { database } from "./plugins/database/database.js";
|
|
38
41
|
import "./plugins/database/index.js";
|
|
39
42
|
import "./plugins/beta-exports.generated.js";
|
|
40
|
-
export { type AgentAdapter, type AgentDefinition, type AgentEvent, type AgentInput, type AgentRunContext, type AgentTool, type AgentToolDefinition, type AgentTools, type AgentToolsFn, type AgentsPluginConfig, AppKitMcpClient, AssertionHandle, AssertionResult, Assessment, type AutoInheritToolsConfig, type BaseSystemPromptOption, CustomJudgeSpec, type DatabaseExports, DatabricksAdapter, DatabricksAuth, DiscoveredEval, DriveResult, EvalDefinition, EvalDriver, EvalProgress, EvalResult, EvalRunSummary, EvalSummary, type FunctionTool, type GenerationParams, type HostedSupervisorTool, type HostedTool, HttpDriverOptions, type IAiSearchConfig, type IDatabaseConfig, type IndexConfig, JudgeConfig, JudgeScore, MatchResult, Matcher, type McpConnectAllResult, type Message, MlflowClient, type PluginToolkitProvider, type Plugins, PostResult, type PromptContext, type RegisteredAgent, ReportOutcome, type RerankerConfig, ResolveDatabricksAuthOptions, type ResolvedToolEntry, type RunAgentInput, type RunAgentResult, RunEvalOptions, RunEvalsOptions, SUPERVISOR_EXTENSION_KEY, type Schema, type SearchFilters, type SearchRequest, type SearchResponse, type SearchResult, Severity, SupervisorApiAdapter, type SupervisorApiAdapterOptions, type SupervisorExtension, type SupervisorTool, TestContext, type Thread, type ThreadStore, type ToolAnnotations, type ToolConfig, type ToolEntry, type ToolProvider, type ToolRegistry, type ToolkitEntry, type ToolkitOptions, type WorkspaceClientLike, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineSchema, defineTool, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, runAgent, runEval, runEvalsInDir, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
|
|
43
|
+
export { type AgentAdapter, type AgentDefinition, type AgentEvent, type AgentInput, type AgentRunContext, type AgentTool, type AgentToolDefinition, type AgentTools, type AgentToolsFn, type AgentsPluginConfig, AppKitMcpClient, AssertionHandle, AssertionResult, Assessment, type AutoInheritToolsConfig, type BaseSystemPromptOption, CustomJudgeSpec, type DatabaseApiConfig, type DatabaseApiWriteOperation, type DatabaseApiWritesConfig, type DatabaseExports, DatabricksAdapter, DatabricksAuth, DatasetRow, DiscoveredEval, DriveResult, type EntityHooks, type EntityMutationHooks, EvalDefinition, EvalDriver, EvalProgress, EvalResult, EvalRunSummary, EvalSummary, type FunctionTool, type GenerationParams, type HookApp, type HookContext, type HostedSupervisorTool, type HostedTool, HttpDriverOptions, type IAiSearchConfig, type IDatabaseConfig, type IndexConfig, JudgeConfig, JudgeScore, MatchResult, Matcher, type McpConnectAllResult, type Message, MlflowClient, type PluginToolkitProvider, type Plugins, PostResult, type PromptContext, ReadEvalDatasetOptions, type ReadSerializer, type ReadSerializerContext, type RegisteredAgent, ReportOutcome, type RerankerConfig, ResolveDatabricksAuthOptions, type ResolvedToolEntry, type RunAgentInput, type RunAgentResult, RunEvalOptions, RunEvalsOptions, SUPERVISOR_EXTENSION_KEY, type Schema, type SearchFilters, type SearchRequest, type SearchResponse, type SearchResult, Severity, SupervisorApiAdapter, type SupervisorApiAdapterOptions, type SupervisorExtension, type SupervisorTool, TestContext, type Thread, type ThreadStore, type ToolAnnotations, type ToolConfig, type ToolEntry, type ToolProvider, type ToolRegistry, type ToolkitEntry, type ToolkitOptions, type TransactionClient, type WorkspaceClientLike, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineSchema, defineTool, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, readEvalDataset, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, resolveWorkspaceClient, runAgent, runEval, runEvalsInDir, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
|
package/dist/beta.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { AppKitMcpClient } from "./connectors/mcp/client.js";
|
|
2
|
-
import { resolveDatabricksAuth } from "./connectors/mlflow/auth.js";
|
|
2
|
+
import { resolveDatabricksAuth, resolveWorkspaceClient } from "./connectors/mlflow/auth.js";
|
|
3
3
|
import { MlflowClient, normalizeHost } from "./connectors/mlflow/client.js";
|
|
4
4
|
import { tool } from "./core/agent/tools/tool.js";
|
|
5
5
|
import { defineTool, executeFromRegistry, toolsFromRegistry } from "./core/agent/tools/define-tool.js";
|
|
@@ -15,6 +15,7 @@ import { isToolkitEntry } from "./core/agent/types.js";
|
|
|
15
15
|
import { runAgent } from "./core/agent/run-agent.js";
|
|
16
16
|
import "./core/agent/tools/index.js";
|
|
17
17
|
import "./database/schema-builder/index.js";
|
|
18
|
+
import { readEvalDataset } from "./evals/dataset.js";
|
|
18
19
|
import { defineEval } from "./evals/define-eval.js";
|
|
19
20
|
import { discoverEvalFiles } from "./evals/discover.js";
|
|
20
21
|
import { createHttpDriver } from "./evals/http-driver.js";
|
|
@@ -32,4 +33,4 @@ import { aiSearch } from "./plugins/ai-search/ai-search.js";
|
|
|
32
33
|
import { database } from "./plugins/database/database.js";
|
|
33
34
|
import "./plugins/beta-exports.generated.js";
|
|
34
35
|
|
|
35
|
-
export { AppKitMcpClient, DatabricksAdapter, MlflowClient, SUPERVISOR_EXTENSION_KEY, SupervisorApiAdapter, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineSchema, defineTool, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, runAgent, runEval, runEvalsInDir, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
|
|
36
|
+
export { AppKitMcpClient, DatabricksAdapter, MlflowClient, SUPERVISOR_EXTENSION_KEY, SupervisorApiAdapter, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineSchema, defineTool, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, readEvalDataset, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, resolveWorkspaceClient, runAgent, runEval, runEvalsInDir, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
|
|
@@ -82,6 +82,12 @@ async function runAgentEval(filter, opts) {
|
|
|
82
82
|
host: opts.databricksHost ?? process.env.DATABRICKS_HOST,
|
|
83
83
|
token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN
|
|
84
84
|
}) ?? {};
|
|
85
|
+
const warehouseId = opts.warehouseId ?? process.env.DATABRICKS_WAREHOUSE_ID;
|
|
86
|
+
const workspaceClient = runner.resolveWorkspaceClient({
|
|
87
|
+
profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,
|
|
88
|
+
host: opts.databricksHost ?? process.env.DATABRICKS_HOST,
|
|
89
|
+
token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN
|
|
90
|
+
});
|
|
85
91
|
let summary;
|
|
86
92
|
try {
|
|
87
93
|
summary = await runner.runEvalsInDir({
|
|
@@ -93,6 +99,8 @@ async function runAgentEval(filter, opts) {
|
|
|
93
99
|
concurrency: opts.concurrency,
|
|
94
100
|
mlflow: resolveMlflow(opts, auth),
|
|
95
101
|
judge: resolveJudge(opts, auth),
|
|
102
|
+
workspaceClient,
|
|
103
|
+
warehouseId,
|
|
96
104
|
onEvent: makeProgressReporter(runner, opts.url)
|
|
97
105
|
});
|
|
98
106
|
} catch (err) {
|
|
@@ -105,7 +113,7 @@ async function runAgentEval(filter, opts) {
|
|
|
105
113
|
else console.log("\nMLflow evaluation run skipped — pass --experiment (or set MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.");
|
|
106
114
|
if (!runner.summarize(summary.results).allPassed) process.exitCode = 1;
|
|
107
115
|
}
|
|
108
|
-
const agentEvalCommand = new Command("eval").description("Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app").argument("[filter]", "Only run evals whose <agent>/<id> contains this substring (or an exact agent id)").option("--url <url>", "Base URL of the running app", "http://localhost:3000").option("--strict", "Fail on soft-assertion misses too", false).option("--concurrency <n>", "Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)", (v) => Number.parseInt(v, 10)).option("--root <dir>", "Project root containing server/agents/ (default: cwd)").option("--header <header...>", "Extra request header as 'Key: value' (repeatable)").option("--profile <name>", "Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)").option("--databricks-host <host>", "Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)").option("--databricks-token <token>", "Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)").option("--experiment <id>", "MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)").option("--warehouse-id <id>", "SQL warehouse id for writing assessments to UC-backed experiments (default:
|
|
116
|
+
const agentEvalCommand = new Command("eval").description("Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app").argument("[filter]", "Only run evals whose <agent>/<id> contains this substring (or an exact agent id)").option("--url <url>", "Base URL of the running app", "http://localhost:3000").option("--strict", "Fail on soft-assertion misses too", false).option("--concurrency <n>", "Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)", (v) => Number.parseInt(v, 10)).option("--root <dir>", "Project root containing server/agents/ (default: cwd)").option("--header <header...>", "Extra request header as 'Key: value' (repeatable)").option("--profile <name>", "Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)").option("--databricks-host <host>", "Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)").option("--databricks-token <token>", "Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)").option("--experiment <id>", "MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)").option("--warehouse-id <id>", "SQL warehouse id for reading managed eval datasets and writing assessments to UC-backed experiments (default: DATABRICKS_WAREHOUSE_ID, or MLFLOW_TRACING_SQL_WAREHOUSE_ID for assessments)").option("--judge-model <endpoint>", "Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)").action(runAgentEval);
|
|
109
117
|
|
|
110
118
|
//#endregion
|
|
111
119
|
export { agentEvalCommand };
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"eval.js","names":[],"sources":["../../../../src/cli/commands/agent/eval.ts"],"sourcesContent":["import { Command } from \"commander\";\n\ninterface EvalRunSummary {\n results: unknown[];\n mlflow?: {\n runId: string;\n report: {\n written: number;\n skipped: number;\n failures: Array<{ traceId: string; status?: number; error?: string }>;\n };\n finish: { finished: boolean; metricsError?: string; finishError?: string };\n };\n}\n\ntype EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: unknown; index: number; total: number };\n\n/** Subset of `@databricks/appkit/beta`'s eval runner used by this command. */\ninterface EvalRunner {\n runEvalsInDir(opts: {\n rootDir?: string;\n baseUrl: string;\n filter?: string;\n strict?: boolean;\n headers?: Record<string, string>;\n concurrency?: number;\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n sqlWarehouseId?: string;\n };\n judge?: { host: string; token: string; model: string };\n onEvent?: (event: EvalProgress) => void;\n }): Promise<EvalRunSummary>;\n resolveDatabricksAuth(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): Promise<{ host: string; token: string } | undefined>;\n formatEvalHeadline(result: unknown): string;\n formatEvalDetail(result: unknown): string[];\n formatSummaryLine(results: unknown[]): string;\n summarize(results: unknown[]): { allPassed: boolean };\n}\n\n/**\n * Loaded at runtime from the consuming project so this command (which ships in\n * `@databricks/shared`) doesn't take a build-time dependency on appkit. The\n * specifier is a variable so the type checker treats it as `any`.\n */\nasync function loadRunner(): Promise<EvalRunner> {\n const spec = \"@databricks/appkit/beta\";\n try {\n return (await import(spec)) as unknown as EvalRunner;\n } catch (err) {\n throw new Error(\n \"Could not load @databricks/appkit. Run `appkit agent eval` from a \" +\n \"project with @databricks/appkit installed. \" +\n `Cause: ${err instanceof Error ? err.message : String(err)}`,\n );\n }\n}\n\nfunction parseHeaders(values: string[]): Record<string, string> {\n const headers: Record<string, string> = {};\n for (const v of values) {\n const i = v.indexOf(\":\");\n if (i === -1) continue;\n headers[v.slice(0, i).trim()] = v.slice(i + 1).trim();\n }\n return headers;\n}\n\ninterface EvalOptions {\n url: string;\n strict?: boolean;\n root?: string;\n header?: string[];\n profile?: string;\n databricksHost?: string;\n databricksToken?: string;\n experiment?: string;\n judgeModel?: string;\n concurrency?: number;\n warehouseId?: string;\n}\n\n/** Resolved Databricks host + bearer (either field may be absent). */\ntype Auth = { host?: string; token?: string };\n\n/**\n * Native MLflow \"Evaluation run\" config — only when creds + an experiment are\n * all present (traces live in the app; the run + scores are driven from here).\n */\nfunction resolveMlflow(opts: EvalOptions, auth: Auth) {\n const experimentId = opts.experiment ?? process.env.MLFLOW_EXPERIMENT_ID;\n if (!(auth.host && auth.token && experimentId)) return undefined;\n // UC-backed experiments need a SQL warehouse to write assessments to their\n // V4 traces. Mirror mlflow's env var, and accept the common DATABRICKS one.\n const sqlWarehouseId =\n opts.warehouseId ??\n process.env.MLFLOW_TRACING_SQL_WAREHOUSE_ID ??\n process.env.DATABRICKS_WAREHOUSE_ID;\n return {\n host: auth.host,\n token: auth.token,\n experimentId,\n ...(sqlWarehouseId ? { sqlWarehouseId } : {}),\n };\n}\n\n/** LLM-as-judge config — reuses the Databricks creds + a judge serving endpoint. */\nfunction resolveJudge(opts: EvalOptions, auth: Auth) {\n const model = opts.judgeModel ?? process.env.APPKIT_JUDGE_MODEL;\n return model && auth.host && auth.token\n ? { host: auth.host, token: auth.token, model }\n : undefined;\n}\n\n/** Progress reporter: stream each eval as it runs instead of going silent. */\nfunction makeProgressReporter(\n runner: EvalRunner,\n url: string,\n): (event: EvalProgress) => void {\n return (event) => {\n switch (event.type) {\n case \"discovered\":\n console.log(\n `Running ${event.total} eval${event.total === 1 ? \"\" : \"s\"} against ${url}\\n`,\n );\n break;\n case \"run-created\":\n console.log(`MLflow evaluation run: ${event.runId}\\n`);\n break;\n case \"result\": {\n // One full line per completion — evals run concurrently, so a split\n // \"start … glyph\" prefix would interleave into garbage.\n console.log(\n `[${event.index + 1}/${event.total}] ${runner.formatEvalHeadline(event.result)}`,\n );\n for (const line of runner.formatEvalDetail(event.result)) {\n console.log(line);\n }\n break;\n }\n }\n };\n}\n\nfunction formatFailureLine(f: {\n traceId: string;\n status?: number;\n error?: string;\n}): string {\n return ` ✗ trace ${f.traceId}: ${f.status ?? \"\"} ${f.error ?? \"\"}`.trim();\n}\n\n/** Print the MLflow assessment/finish outcome after a run that created one. */\nfunction printMlflowOutcome(\n mlflow: NonNullable<EvalRunSummary[\"mlflow\"]>,\n): void {\n const { report, finish } = mlflow;\n console.log(\n `MLflow: ${report.written} assessment(s) written` +\n (report.skipped ? `, ${report.skipped} skipped` : \"\") +\n (report.failures.length ? `, ${report.failures.length} failed` : \"\"),\n );\n for (const f of report.failures) {\n console.error(formatFailureLine(f));\n }\n if (finish.metricsError) {\n console.error(` ⚠ metrics not logged: ${finish.metricsError}`);\n }\n if (!finish.finished) {\n console.error(\n ` ✗ run left RUNNING — failed to finish: ${finish.finishError ?? \"unknown\"}`,\n );\n }\n}\n\nasync function runAgentEval(\n filter: string | undefined,\n opts: EvalOptions,\n): Promise<void> {\n const runner = await loadRunner();\n\n // Resolve Databricks host + bearer the AppKit-native way: an explicit\n // host/token (or DATABRICKS_* env) wins; otherwise the SDK mints an OAuth\n // token from the CLI profile — so no hand-set PAT is required.\n const auth: Auth =\n (await runner.resolveDatabricksAuth({\n profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,\n host: opts.databricksHost ?? process.env.DATABRICKS_HOST,\n token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN,\n })) ?? {};\n\n let summary: EvalRunSummary;\n try {\n summary = await runner.runEvalsInDir({\n rootDir: opts.root,\n baseUrl: opts.url,\n filter,\n strict: opts.strict,\n headers: opts.header ? parseHeaders(opts.header) : undefined,\n concurrency: opts.concurrency,\n mlflow: resolveMlflow(opts, auth),\n judge: resolveJudge(opts, auth),\n onEvent: makeProgressReporter(runner, opts.url),\n });\n } catch (err) {\n // Setup failures (e.g. a bad --experiment for the MLflow run) reject before\n // any eval runs; surface a clean message + non-zero exit rather than an\n // unhandled promise rejection with a raw stack.\n console.error(\n `\\nEval run failed: ${err instanceof Error ? err.message : String(err)}`,\n );\n process.exitCode = 1;\n return;\n }\n console.log(`\\n${runner.formatSummaryLine(summary.results)}`);\n\n if (summary.mlflow) {\n printMlflowOutcome(summary.mlflow);\n } else {\n console.log(\n \"\\nMLflow evaluation run skipped — pass --experiment (or set\" +\n \" MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.\",\n );\n }\n\n if (!runner.summarize(summary.results).allPassed) {\n process.exitCode = 1;\n }\n}\n\nexport const agentEvalCommand = new Command(\"eval\")\n .description(\n \"Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app\",\n )\n .argument(\n \"[filter]\",\n \"Only run evals whose <agent>/<id> contains this substring (or an exact agent id)\",\n )\n .option(\"--url <url>\", \"Base URL of the running app\", \"http://localhost:3000\")\n .option(\"--strict\", \"Fail on soft-assertion misses too\", false)\n .option(\n \"--concurrency <n>\",\n \"Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)\",\n (v) => Number.parseInt(v, 10),\n )\n .option(\n \"--root <dir>\",\n \"Project root containing server/agents/ (default: cwd)\",\n )\n .option(\n \"--header <header...>\",\n \"Extra request header as 'Key: value' (repeatable)\",\n )\n .option(\n \"--profile <name>\",\n \"Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)\",\n )\n .option(\n \"--databricks-host <host>\",\n \"Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)\",\n )\n .option(\n \"--databricks-token <token>\",\n \"Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)\",\n )\n .option(\n \"--experiment <id>\",\n \"MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)\",\n )\n .option(\n \"--warehouse-id <id>\",\n \"SQL warehouse id for writing assessments to UC-backed experiments (default: MLFLOW_TRACING_SQL_WAREHOUSE_ID or DATABRICKS_WAREHOUSE_ID)\",\n )\n .option(\n \"--judge-model <endpoint>\",\n \"Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)\",\n )\n .action(runAgentEval);\n"],"mappings":";;;;;;;;AAuDA,eAAe,aAAkC;CAC/C,MAAM,OAAO;AACb,KAAI;AACF,SAAQ,MAAM,OAAO;UACd,KAAK;AACZ,QAAM,IAAI,MACR,yHAEY,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAC7D;;;AAIL,SAAS,aAAa,QAA0C;CAC9D,MAAM,UAAkC,EAAE;AAC1C,MAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,IAAI,EAAE,QAAQ,IAAI;AACxB,MAAI,MAAM,GAAI;AACd,UAAQ,EAAE,MAAM,GAAG,EAAE,CAAC,MAAM,IAAI,EAAE,MAAM,IAAI,EAAE,CAAC,MAAM;;AAEvD,QAAO;;;;;;AAwBT,SAAS,cAAc,MAAmB,MAAY;CACpD,MAAM,eAAe,KAAK,cAAc,QAAQ,IAAI;AACpD,KAAI,EAAE,KAAK,QAAQ,KAAK,SAAS,cAAe,QAAO;CAGvD,MAAM,iBACJ,KAAK,eACL,QAAQ,IAAI,mCACZ,QAAQ,IAAI;AACd,QAAO;EACL,MAAM,KAAK;EACX,OAAO,KAAK;EACZ;EACA,GAAI,iBAAiB,EAAE,gBAAgB,GAAG,EAAE;EAC7C;;;AAIH,SAAS,aAAa,MAAmB,MAAY;CACnD,MAAM,QAAQ,KAAK,cAAc,QAAQ,IAAI;AAC7C,QAAO,SAAS,KAAK,QAAQ,KAAK,QAC9B;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;EAAO;EAAO,GAC7C;;;AAIN,SAAS,qBACP,QACA,KAC+B;AAC/B,SAAQ,UAAU;AAChB,UAAQ,MAAM,MAAd;GACE,KAAK;AACH,YAAQ,IACN,WAAW,MAAM,MAAM,OAAO,MAAM,UAAU,IAAI,KAAK,IAAI,WAAW,IAAI,IAC3E;AACD;GACF,KAAK;AACH,YAAQ,IAAI,0BAA0B,MAAM,MAAM,IAAI;AACtD;GACF,KAAK;AAGH,YAAQ,IACN,IAAI,MAAM,QAAQ,EAAE,GAAG,MAAM,MAAM,IAAI,OAAO,mBAAmB,MAAM,OAAO,GAC/E;AACD,SAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,OAAO,CACtD,SAAQ,IAAI,KAAK;AAEnB;;;;AAMR,SAAS,kBAAkB,GAIhB;AACT,QAAO,aAAa,EAAE,QAAQ,IAAI,EAAE,UAAU,GAAG,GAAG,EAAE,SAAS,KAAK,MAAM;;;AAI5E,SAAS,mBACP,QACM;CACN,MAAM,EAAE,QAAQ,WAAW;AAC3B,SAAQ,IACN,WAAW,OAAO,QAAQ,2BACvB,OAAO,UAAU,KAAK,OAAO,QAAQ,YAAY,OACjD,OAAO,SAAS,SAAS,KAAK,OAAO,SAAS,OAAO,WAAW,IACpE;AACD,MAAK,MAAM,KAAK,OAAO,SACrB,SAAQ,MAAM,kBAAkB,EAAE,CAAC;AAErC,KAAI,OAAO,aACT,SAAQ,MAAM,2BAA2B,OAAO,eAAe;AAEjE,KAAI,CAAC,OAAO,SACV,SAAQ,MACN,4CAA4C,OAAO,eAAe,YACnE;;AAIL,eAAe,aACb,QACA,MACe;CACf,MAAM,SAAS,MAAM,YAAY;CAKjC,MAAM,OACH,MAAM,OAAO,sBAAsB;EAClC,SAAS,KAAK,WAAW,QAAQ,IAAI;EACrC,MAAM,KAAK,kBAAkB,QAAQ,IAAI;EACzC,OAAO,KAAK,mBAAmB,QAAQ,IAAI;EAC5C,CAAC,IAAK,EAAE;CAEX,IAAI;AACJ,KAAI;AACF,YAAU,MAAM,OAAO,cAAc;GACnC,SAAS,KAAK;GACd,SAAS,KAAK;GACd;GACA,QAAQ,KAAK;GACb,SAAS,KAAK,SAAS,aAAa,KAAK,OAAO,GAAG;GACnD,aAAa,KAAK;GAClB,QAAQ,cAAc,MAAM,KAAK;GACjC,OAAO,aAAa,MAAM,KAAK;GAC/B,SAAS,qBAAqB,QAAQ,KAAK,IAAI;GAChD,CAAC;UACK,KAAK;AAIZ,UAAQ,MACN,sBAAsB,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GACvE;AACD,UAAQ,WAAW;AACnB;;AAEF,SAAQ,IAAI,KAAK,OAAO,kBAAkB,QAAQ,QAAQ,GAAG;AAE7D,KAAI,QAAQ,OACV,oBAAmB,QAAQ,OAAO;KAElC,SAAQ,IACN,oIAED;AAGH,KAAI,CAAC,OAAO,UAAU,QAAQ,QAAQ,CAAC,UACrC,SAAQ,WAAW;;AAIvB,MAAa,mBAAmB,IAAI,QAAQ,OAAO,CAChD,YACC,6EACD,CACA,SACC,YACA,mFACD,CACA,OAAO,eAAe,+BAA+B,wBAAwB,CAC7E,OAAO,YAAY,qCAAqC,MAAM,CAC9D,OACC,qBACA,0GACC,MAAM,OAAO,SAAS,GAAG,GAAG,CAC9B,CACA,OACC,gBACA,wDACD,CACA,OACC,wBACA,oDACD,CACA,OACC,oBACA,6FACD,CACA,OACC,4BACA,4EACD,CACA,OACC,8BACA,8EACD,CACA,OACC,qBACA,8EACD,CACA,OACC,uBACA,0IACD,CACA,OACC,4BACA,kGACD,CACA,OAAO,aAAa"}
|
|
1
|
+
{"version":3,"file":"eval.js","names":[],"sources":["../../../../src/cli/commands/agent/eval.ts"],"sourcesContent":["import { Command } from \"commander\";\n\ninterface EvalRunSummary {\n results: unknown[];\n mlflow?: {\n runId: string;\n report: {\n written: number;\n skipped: number;\n failures: Array<{ traceId: string; status?: number; error?: string }>;\n };\n finish: { finished: boolean; metricsError?: string; finishError?: string };\n };\n}\n\ntype EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: unknown; index: number; total: number };\n\n/** Subset of `@databricks/appkit/beta`'s eval runner used by this command. */\ninterface EvalRunner {\n runEvalsInDir(opts: {\n rootDir?: string;\n baseUrl: string;\n filter?: string;\n strict?: boolean;\n headers?: Record<string, string>;\n concurrency?: number;\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n sqlWarehouseId?: string;\n };\n judge?: { host: string; token: string; model: string };\n workspaceClient?: unknown;\n warehouseId?: string;\n onEvent?: (event: EvalProgress) => void;\n }): Promise<EvalRunSummary>;\n resolveDatabricksAuth(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): Promise<{ host: string; token: string } | undefined>;\n resolveWorkspaceClient(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): unknown;\n formatEvalHeadline(result: unknown): string;\n evalGlyph(result: unknown): string;\n formatEvalDetail(result: unknown): string[];\n formatSummaryLine(results: unknown[]): string;\n summarize(results: unknown[]): { allPassed: boolean };\n}\n\n/**\n * Loaded at runtime from the consuming project so this command (which ships in\n * `@databricks/shared`) doesn't take a build-time dependency on appkit. The\n * specifier is a variable so the type checker treats it as `any`.\n */\nasync function loadRunner(): Promise<EvalRunner> {\n const spec = \"@databricks/appkit/beta\";\n try {\n return (await import(spec)) as unknown as EvalRunner;\n } catch (err) {\n throw new Error(\n \"Could not load @databricks/appkit. Run `appkit agent eval` from a \" +\n \"project with @databricks/appkit installed. \" +\n `Cause: ${err instanceof Error ? err.message : String(err)}`,\n );\n }\n}\n\nfunction parseHeaders(values: string[]): Record<string, string> {\n const headers: Record<string, string> = {};\n for (const v of values) {\n const i = v.indexOf(\":\");\n if (i === -1) continue;\n headers[v.slice(0, i).trim()] = v.slice(i + 1).trim();\n }\n return headers;\n}\n\ninterface EvalOptions {\n url: string;\n strict?: boolean;\n root?: string;\n header?: string[];\n profile?: string;\n databricksHost?: string;\n databricksToken?: string;\n experiment?: string;\n judgeModel?: string;\n concurrency?: number;\n warehouseId?: string;\n}\n\n/** Resolved Databricks host + bearer (either field may be absent). */\ntype Auth = { host?: string; token?: string };\n\n/**\n * Native MLflow \"Evaluation run\" config — only when creds + an experiment are\n * all present (traces live in the app; the run + scores are driven from here).\n */\nfunction resolveMlflow(opts: EvalOptions, auth: Auth) {\n const experimentId = opts.experiment ?? process.env.MLFLOW_EXPERIMENT_ID;\n if (!(auth.host && auth.token && experimentId)) return undefined;\n // UC-backed experiments need a SQL warehouse to write assessments to their\n // V4 traces. Mirror mlflow's env var, and accept the common DATABRICKS one.\n const sqlWarehouseId =\n opts.warehouseId ??\n process.env.MLFLOW_TRACING_SQL_WAREHOUSE_ID ??\n process.env.DATABRICKS_WAREHOUSE_ID;\n return {\n host: auth.host,\n token: auth.token,\n experimentId,\n ...(sqlWarehouseId ? { sqlWarehouseId } : {}),\n };\n}\n\n/** LLM-as-judge config — reuses the Databricks creds + a judge serving endpoint. */\nfunction resolveJudge(opts: EvalOptions, auth: Auth) {\n const model = opts.judgeModel ?? process.env.APPKIT_JUDGE_MODEL;\n return model && auth.host && auth.token\n ? { host: auth.host, token: auth.token, model }\n : undefined;\n}\n\n/** Progress reporter: stream each eval as it runs instead of going silent. */\nfunction makeProgressReporter(\n runner: EvalRunner,\n url: string,\n): (event: EvalProgress) => void {\n return (event) => {\n switch (event.type) {\n case \"discovered\":\n console.log(\n `Running ${event.total} eval${event.total === 1 ? \"\" : \"s\"} against ${url}\\n`,\n );\n break;\n case \"run-created\":\n console.log(`MLflow evaluation run: ${event.runId}\\n`);\n break;\n case \"result\": {\n // One full line per completion — evals run concurrently, so a split\n // \"start … glyph\" prefix would interleave into garbage.\n console.log(\n `[${event.index + 1}/${event.total}] ${runner.formatEvalHeadline(event.result)}`,\n );\n for (const line of runner.formatEvalDetail(event.result)) {\n console.log(line);\n }\n break;\n }\n }\n };\n}\n\nfunction formatFailureLine(f: {\n traceId: string;\n status?: number;\n error?: string;\n}): string {\n return ` ✗ trace ${f.traceId}: ${f.status ?? \"\"} ${f.error ?? \"\"}`.trim();\n}\n\n/** Print the MLflow assessment/finish outcome after a run that created one. */\nfunction printMlflowOutcome(\n mlflow: NonNullable<EvalRunSummary[\"mlflow\"]>,\n): void {\n const { report, finish } = mlflow;\n console.log(\n `MLflow: ${report.written} assessment(s) written` +\n (report.skipped ? `, ${report.skipped} skipped` : \"\") +\n (report.failures.length ? `, ${report.failures.length} failed` : \"\"),\n );\n for (const f of report.failures) {\n console.error(formatFailureLine(f));\n }\n if (finish.metricsError) {\n console.error(` ⚠ metrics not logged: ${finish.metricsError}`);\n }\n if (!finish.finished) {\n console.error(\n ` ✗ run left RUNNING — failed to finish: ${finish.finishError ?? \"unknown\"}`,\n );\n }\n}\n\nasync function runAgentEval(\n filter: string | undefined,\n opts: EvalOptions,\n): Promise<void> {\n const runner = await loadRunner();\n\n // Resolve Databricks host + bearer the AppKit-native way: an explicit\n // host/token (or DATABRICKS_* env) wins; otherwise the SDK mints an OAuth\n // token from the CLI profile — so no hand-set PAT is required.\n const auth: Auth =\n (await runner.resolveDatabricksAuth({\n profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,\n host: opts.databricksHost ?? process.env.DATABRICKS_HOST,\n token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN,\n })) ?? {};\n\n // Managed-dataset reads: a workspace client (same profile/host/token) + a SQL\n // warehouse. Only needed by evals that declare `dataset`.\n const warehouseId = opts.warehouseId ?? process.env.DATABRICKS_WAREHOUSE_ID;\n const workspaceClient = runner.resolveWorkspaceClient({\n profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,\n host: opts.databricksHost ?? process.env.DATABRICKS_HOST,\n token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN,\n });\n\n let summary: EvalRunSummary;\n try {\n summary = await runner.runEvalsInDir({\n rootDir: opts.root,\n baseUrl: opts.url,\n filter,\n strict: opts.strict,\n headers: opts.header ? parseHeaders(opts.header) : undefined,\n concurrency: opts.concurrency,\n mlflow: resolveMlflow(opts, auth),\n judge: resolveJudge(opts, auth),\n workspaceClient,\n warehouseId,\n onEvent: makeProgressReporter(runner, opts.url),\n });\n } catch (err) {\n // Setup failures (e.g. a bad --experiment for the MLflow run) reject before\n // any eval runs; surface a clean message + non-zero exit rather than an\n // unhandled promise rejection with a raw stack.\n console.error(\n `\\nEval run failed: ${err instanceof Error ? err.message : String(err)}`,\n );\n process.exitCode = 1;\n return;\n }\n console.log(`\\n${runner.formatSummaryLine(summary.results)}`);\n\n if (summary.mlflow) {\n printMlflowOutcome(summary.mlflow);\n } else {\n console.log(\n \"\\nMLflow evaluation run skipped — pass --experiment (or set\" +\n \" MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.\",\n );\n }\n\n if (!runner.summarize(summary.results).allPassed) {\n process.exitCode = 1;\n }\n}\n\nexport const agentEvalCommand = new Command(\"eval\")\n .description(\n \"Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app\",\n )\n .argument(\n \"[filter]\",\n \"Only run evals whose <agent>/<id> contains this substring (or an exact agent id)\",\n )\n .option(\"--url <url>\", \"Base URL of the running app\", \"http://localhost:3000\")\n .option(\"--strict\", \"Fail on soft-assertion misses too\", false)\n .option(\n \"--concurrency <n>\",\n \"Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)\",\n (v) => Number.parseInt(v, 10),\n )\n .option(\n \"--root <dir>\",\n \"Project root containing server/agents/ (default: cwd)\",\n )\n .option(\n \"--header <header...>\",\n \"Extra request header as 'Key: value' (repeatable)\",\n )\n .option(\n \"--profile <name>\",\n \"Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)\",\n )\n .option(\n \"--databricks-host <host>\",\n \"Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)\",\n )\n .option(\n \"--databricks-token <token>\",\n \"Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)\",\n )\n .option(\n \"--experiment <id>\",\n \"MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)\",\n )\n .option(\n \"--warehouse-id <id>\",\n \"SQL warehouse id for reading managed eval datasets and writing assessments to UC-backed experiments (default: DATABRICKS_WAREHOUSE_ID, or MLFLOW_TRACING_SQL_WAREHOUSE_ID for assessments)\",\n )\n .option(\n \"--judge-model <endpoint>\",\n \"Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)\",\n )\n .action(runAgentEval);\n"],"mappings":";;;;;;;;AA+DA,eAAe,aAAkC;CAC/C,MAAM,OAAO;AACb,KAAI;AACF,SAAQ,MAAM,OAAO;UACd,KAAK;AACZ,QAAM,IAAI,MACR,yHAEY,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAC7D;;;AAIL,SAAS,aAAa,QAA0C;CAC9D,MAAM,UAAkC,EAAE;AAC1C,MAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,IAAI,EAAE,QAAQ,IAAI;AACxB,MAAI,MAAM,GAAI;AACd,UAAQ,EAAE,MAAM,GAAG,EAAE,CAAC,MAAM,IAAI,EAAE,MAAM,IAAI,EAAE,CAAC,MAAM;;AAEvD,QAAO;;;;;;AAwBT,SAAS,cAAc,MAAmB,MAAY;CACpD,MAAM,eAAe,KAAK,cAAc,QAAQ,IAAI;AACpD,KAAI,EAAE,KAAK,QAAQ,KAAK,SAAS,cAAe,QAAO;CAGvD,MAAM,iBACJ,KAAK,eACL,QAAQ,IAAI,mCACZ,QAAQ,IAAI;AACd,QAAO;EACL,MAAM,KAAK;EACX,OAAO,KAAK;EACZ;EACA,GAAI,iBAAiB,EAAE,gBAAgB,GAAG,EAAE;EAC7C;;;AAIH,SAAS,aAAa,MAAmB,MAAY;CACnD,MAAM,QAAQ,KAAK,cAAc,QAAQ,IAAI;AAC7C,QAAO,SAAS,KAAK,QAAQ,KAAK,QAC9B;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;EAAO;EAAO,GAC7C;;;AAIN,SAAS,qBACP,QACA,KAC+B;AAC/B,SAAQ,UAAU;AAChB,UAAQ,MAAM,MAAd;GACE,KAAK;AACH,YAAQ,IACN,WAAW,MAAM,MAAM,OAAO,MAAM,UAAU,IAAI,KAAK,IAAI,WAAW,IAAI,IAC3E;AACD;GACF,KAAK;AACH,YAAQ,IAAI,0BAA0B,MAAM,MAAM,IAAI;AACtD;GACF,KAAK;AAGH,YAAQ,IACN,IAAI,MAAM,QAAQ,EAAE,GAAG,MAAM,MAAM,IAAI,OAAO,mBAAmB,MAAM,OAAO,GAC/E;AACD,SAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,OAAO,CACtD,SAAQ,IAAI,KAAK;AAEnB;;;;AAMR,SAAS,kBAAkB,GAIhB;AACT,QAAO,aAAa,EAAE,QAAQ,IAAI,EAAE,UAAU,GAAG,GAAG,EAAE,SAAS,KAAK,MAAM;;;AAI5E,SAAS,mBACP,QACM;CACN,MAAM,EAAE,QAAQ,WAAW;AAC3B,SAAQ,IACN,WAAW,OAAO,QAAQ,2BACvB,OAAO,UAAU,KAAK,OAAO,QAAQ,YAAY,OACjD,OAAO,SAAS,SAAS,KAAK,OAAO,SAAS,OAAO,WAAW,IACpE;AACD,MAAK,MAAM,KAAK,OAAO,SACrB,SAAQ,MAAM,kBAAkB,EAAE,CAAC;AAErC,KAAI,OAAO,aACT,SAAQ,MAAM,2BAA2B,OAAO,eAAe;AAEjE,KAAI,CAAC,OAAO,SACV,SAAQ,MACN,4CAA4C,OAAO,eAAe,YACnE;;AAIL,eAAe,aACb,QACA,MACe;CACf,MAAM,SAAS,MAAM,YAAY;CAKjC,MAAM,OACH,MAAM,OAAO,sBAAsB;EAClC,SAAS,KAAK,WAAW,QAAQ,IAAI;EACrC,MAAM,KAAK,kBAAkB,QAAQ,IAAI;EACzC,OAAO,KAAK,mBAAmB,QAAQ,IAAI;EAC5C,CAAC,IAAK,EAAE;CAIX,MAAM,cAAc,KAAK,eAAe,QAAQ,IAAI;CACpD,MAAM,kBAAkB,OAAO,uBAAuB;EACpD,SAAS,KAAK,WAAW,QAAQ,IAAI;EACrC,MAAM,KAAK,kBAAkB,QAAQ,IAAI;EACzC,OAAO,KAAK,mBAAmB,QAAQ,IAAI;EAC5C,CAAC;CAEF,IAAI;AACJ,KAAI;AACF,YAAU,MAAM,OAAO,cAAc;GACnC,SAAS,KAAK;GACd,SAAS,KAAK;GACd;GACA,QAAQ,KAAK;GACb,SAAS,KAAK,SAAS,aAAa,KAAK,OAAO,GAAG;GACnD,aAAa,KAAK;GAClB,QAAQ,cAAc,MAAM,KAAK;GACjC,OAAO,aAAa,MAAM,KAAK;GAC/B;GACA;GACA,SAAS,qBAAqB,QAAQ,KAAK,IAAI;GAChD,CAAC;UACK,KAAK;AAIZ,UAAQ,MACN,sBAAsB,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GACvE;AACD,UAAQ,WAAW;AACnB;;AAEF,SAAQ,IAAI,KAAK,OAAO,kBAAkB,QAAQ,QAAQ,GAAG;AAE7D,KAAI,QAAQ,OACV,oBAAmB,QAAQ,OAAO;KAElC,SAAQ,IACN,oIAED;AAGH,KAAI,CAAC,OAAO,UAAU,QAAQ,QAAQ,CAAC,UACrC,SAAQ,WAAW;;AAIvB,MAAa,mBAAmB,IAAI,QAAQ,OAAO,CAChD,YACC,6EACD,CACA,SACC,YACA,mFACD,CACA,OAAO,eAAe,+BAA+B,wBAAwB,CAC7E,OAAO,YAAY,qCAAqC,MAAM,CAC9D,OACC,qBACA,0GACC,MAAM,OAAO,SAAS,GAAG,GAAG,CAC9B,CACA,OACC,gBACA,wDACD,CACA,OACC,wBACA,oDACD,CACA,OACC,oBACA,6FACD,CACA,OACC,4BACA,4EACD,CACA,OACC,8BACA,8EACD,CACA,OACC,qBACA,8EACD,CACA,OACC,uBACA,6LACD,CACA,OACC,4BACA,kGACD,CACA,OAAO,aAAa"}
|
package/dist/connectors/index.js
CHANGED
|
@@ -13,7 +13,7 @@ import "./jobs/index.js";
|
|
|
13
13
|
import { buildMcpHostPolicy } from "./mcp/host-policy.js";
|
|
14
14
|
import { AppKitMcpClient } from "./mcp/client.js";
|
|
15
15
|
import "./mcp/index.js";
|
|
16
|
-
import { resolveDatabricksAuth } from "./mlflow/auth.js";
|
|
16
|
+
import { resolveDatabricksAuth, resolveWorkspaceClient } from "./mlflow/auth.js";
|
|
17
17
|
import { MlflowClient, normalizeHost } from "./mlflow/client.js";
|
|
18
18
|
import { DEFAULT_WAREHOUSE_STARTUP_TIMEOUT_MS, SQLWarehouseConnector } from "./sql-warehouse/client.js";
|
|
19
19
|
import "./sql-warehouse/index.js";
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { WorkspaceClient } from "../../workspace-client/index.js";
|
|
2
|
+
|
|
1
3
|
//#region src/connectors/mlflow/auth.d.ts
|
|
2
4
|
/** Resolved Databricks host + bearer token for the eval runner's REST calls. */
|
|
3
5
|
interface DatabricksAuth {
|
|
@@ -13,6 +15,14 @@ interface ResolveDatabricksAuthOptions {
|
|
|
13
15
|
token?: string;
|
|
14
16
|
}
|
|
15
17
|
declare function resolveDatabricksAuth(options?: ResolveDatabricksAuthOptions): Promise<DatabricksAuth | undefined>;
|
|
18
|
+
/**
|
|
19
|
+
* Construct a Databricks `WorkspaceClient` for the eval runner — the object the
|
|
20
|
+
* SDK-backed connectors (e.g. `SQLWarehouseConnector`) take. An explicit
|
|
21
|
+
* host+token builds a PAT client; otherwise the profile (or ambient config) is
|
|
22
|
+
* used and the SDK resolves credentials, minting OAuth as needed. Returns
|
|
23
|
+
* `undefined` if construction throws (missing/invalid config).
|
|
24
|
+
*/
|
|
25
|
+
declare function resolveWorkspaceClient(options?: ResolveDatabricksAuthOptions): WorkspaceClient | undefined;
|
|
16
26
|
//#endregion
|
|
17
|
-
export { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth };
|
|
27
|
+
export { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth, resolveWorkspaceClient };
|
|
18
28
|
//# sourceMappingURL=auth.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"auth.d.ts","names":[],"sources":["../../../src/connectors/mlflow/auth.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"auth.d.ts","names":[],"sources":["../../../src/connectors/mlflow/auth.ts"],"mappings":";;;;UAMiB,cAAA;EACf,IAAA;EACA,KAAA;AAAA;AAAA,UAGe,4BAAA;EAHV;EAKL,OAAA;EAF2C;EAI3C,IAAA;EAJ2C;EAM3C,KAAA;AAAA;AAAA,iBA8CoB,qBAAA,CACpB,OAAA,GAAS,4BAAA,GACR,OAAA,CAAQ,cAAA;;;AAFX;;;;;iBAiBgB,sBAAA,CACd,OAAA,GAAS,4BAAA,GACR,eAAA"}
|
|
@@ -23,7 +23,8 @@ function extractBearer(headers) {
|
|
|
23
23
|
*/
|
|
24
24
|
async function resolveViaSdk(options) {
|
|
25
25
|
try {
|
|
26
|
-
const client =
|
|
26
|
+
const client = resolveWorkspaceClient(options);
|
|
27
|
+
if (!client) return void 0;
|
|
27
28
|
const headers = new Headers();
|
|
28
29
|
await client.config.authenticate(headers);
|
|
29
30
|
const token = options.token ?? extractBearer(headers);
|
|
@@ -44,7 +45,26 @@ async function resolveDatabricksAuth(options = {}) {
|
|
|
44
45
|
};
|
|
45
46
|
return resolveViaSdk(options);
|
|
46
47
|
}
|
|
48
|
+
/**
|
|
49
|
+
* Construct a Databricks `WorkspaceClient` for the eval runner — the object the
|
|
50
|
+
* SDK-backed connectors (e.g. `SQLWarehouseConnector`) take. An explicit
|
|
51
|
+
* host+token builds a PAT client; otherwise the profile (or ambient config) is
|
|
52
|
+
* used and the SDK resolves credentials, minting OAuth as needed. Returns
|
|
53
|
+
* `undefined` if construction throws (missing/invalid config).
|
|
54
|
+
*/
|
|
55
|
+
function resolveWorkspaceClient(options = {}) {
|
|
56
|
+
try {
|
|
57
|
+
if (options.host && options.token) return createWorkspaceClient({
|
|
58
|
+
host: options.host,
|
|
59
|
+
token: options.token,
|
|
60
|
+
authType: "pat"
|
|
61
|
+
});
|
|
62
|
+
return createWorkspaceClient(options.profile ? { profile: options.profile } : {});
|
|
63
|
+
} catch {
|
|
64
|
+
return;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
47
67
|
|
|
48
68
|
//#endregion
|
|
49
|
-
export { resolveDatabricksAuth };
|
|
69
|
+
export { resolveDatabricksAuth, resolveWorkspaceClient };
|
|
50
70
|
//# sourceMappingURL=auth.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"auth.js","names":[],"sources":["../../../src/connectors/mlflow/auth.ts"],"sourcesContent":["import {
|
|
1
|
+
{"version":3,"file":"auth.js","names":[],"sources":["../../../src/connectors/mlflow/auth.ts"],"sourcesContent":["import {\n createWorkspaceClient,\n type WorkspaceClient,\n} from \"../../workspace-client\";\n\n/** Resolved Databricks host + bearer token for the eval runner's REST calls. */\nexport interface DatabricksAuth {\n host: string;\n token: string;\n}\n\nexport interface ResolveDatabricksAuthOptions {\n /** `~/.databrickscfg` profile to authenticate with (e.g. `dogfood`). */\n profile?: string;\n /** Explicit host; wins over the profile/SDK-resolved host when set. */\n host?: string;\n /** Explicit bearer token; when set, no OAuth is minted (PAT/CI path). */\n token?: string;\n}\n\n/**\n * Resolve `{host, token}` for the eval runner the same way the rest of AppKit\n * authenticates: construct a Databricks `WorkspaceClient` and let its config\n * mint (and later refresh) an OAuth bearer from the CLI profile — no hand-set\n * PAT required. An explicit host/token still wins (PAT or CI env), so the SDK\n * is only consulted for whatever isn't supplied.\n *\n * Returns `undefined` when neither an explicit token nor a resolvable profile\n * yields a bearer, so the caller can treat auth as simply unavailable.\n */\n/** Pull the bearer out of an `Authorization: Bearer <token>` header. */\nfunction extractBearer(headers: Headers): string | undefined {\n return headers.get(\"authorization\")?.replace(/^Bearer\\s+/i, \"\");\n}\n\n/**\n * Resolve `{host, token}` via the SDK: construct a `WorkspaceClient`, let it\n * mint/refresh an OAuth bearer from the profile (or reuse a PAT), and fall back\n * to any explicit host/token the caller supplied. Returns `undefined` when\n * either is missing or the SDK can't resolve credentials.\n */\nasync function resolveViaSdk(\n options: ResolveDatabricksAuthOptions,\n): Promise<DatabricksAuth | undefined> {\n try {\n const client = resolveWorkspaceClient(options);\n if (!client) return undefined;\n const headers = new Headers();\n // Mints the OAuth access token (or reuses a PAT from the profile) and adds\n // an `Authorization: Bearer <token>` header — the same call the connectors\n // use before each request.\n await client.config.authenticate(headers);\n const token = options.token ?? extractBearer(headers);\n const host =\n options.host ??\n (await client.config.getHost()).toString().replace(/\\/+$/, \"\");\n if (!token || !host) return undefined;\n return { host, token };\n } catch {\n return undefined;\n }\n}\n\nexport async function resolveDatabricksAuth(\n options: ResolveDatabricksAuthOptions = {},\n): Promise<DatabricksAuth | undefined> {\n // Fully explicit — no need to touch the SDK.\n if (options.host && options.token) {\n return { host: options.host, token: options.token };\n }\n return resolveViaSdk(options);\n}\n\n/**\n * Construct a Databricks `WorkspaceClient` for the eval runner — the object the\n * SDK-backed connectors (e.g. `SQLWarehouseConnector`) take. An explicit\n * host+token builds a PAT client; otherwise the profile (or ambient config) is\n * used and the SDK resolves credentials, minting OAuth as needed. Returns\n * `undefined` if construction throws (missing/invalid config).\n */\nexport function resolveWorkspaceClient(\n options: ResolveDatabricksAuthOptions = {},\n): WorkspaceClient | undefined {\n try {\n if (options.host && options.token) {\n return createWorkspaceClient({\n host: options.host,\n token: options.token,\n authType: \"pat\",\n });\n }\n return createWorkspaceClient(\n options.profile ? { profile: options.profile } : {},\n );\n } catch {\n return undefined;\n }\n}\n"],"mappings":";;;;;;;;;;;;;;AA+BA,SAAS,cAAc,SAAsC;AAC3D,QAAO,QAAQ,IAAI,gBAAgB,EAAE,QAAQ,eAAe,GAAG;;;;;;;;AASjE,eAAe,cACb,SACqC;AACrC,KAAI;EACF,MAAM,SAAS,uBAAuB,QAAQ;AAC9C,MAAI,CAAC,OAAQ,QAAO;EACpB,MAAM,UAAU,IAAI,SAAS;AAI7B,QAAM,OAAO,OAAO,aAAa,QAAQ;EACzC,MAAM,QAAQ,QAAQ,SAAS,cAAc,QAAQ;EACrD,MAAM,OACJ,QAAQ,SACP,MAAM,OAAO,OAAO,SAAS,EAAE,UAAU,CAAC,QAAQ,QAAQ,GAAG;AAChE,MAAI,CAAC,SAAS,CAAC,KAAM,QAAO;AAC5B,SAAO;GAAE;GAAM;GAAO;SAChB;AACN;;;AAIJ,eAAsB,sBACpB,UAAwC,EAAE,EACL;AAErC,KAAI,QAAQ,QAAQ,QAAQ,MAC1B,QAAO;EAAE,MAAM,QAAQ;EAAM,OAAO,QAAQ;EAAO;AAErD,QAAO,cAAc,QAAQ;;;;;;;;;AAU/B,SAAgB,uBACd,UAAwC,EAAE,EACb;AAC7B,KAAI;AACF,MAAI,QAAQ,QAAQ,QAAQ,MAC1B,QAAO,sBAAsB;GAC3B,MAAM,QAAQ;GACd,OAAO,QAAQ;GACf,UAAU;GACX,CAAC;AAEJ,SAAO,sBACL,QAAQ,UAAU,EAAE,SAAS,QAAQ,SAAS,GAAG,EAAE,CACpD;SACK;AACN"}
|
package/dist/database/errors.js
CHANGED
|
@@ -9,6 +9,10 @@ const definitions = {
|
|
|
9
9
|
message: "Invalid database request",
|
|
10
10
|
statusCode: 400
|
|
11
11
|
},
|
|
12
|
+
VALIDATION_FAILED: {
|
|
13
|
+
message: "Database request failed validation",
|
|
14
|
+
statusCode: 422
|
|
15
|
+
},
|
|
12
16
|
NOT_FOUND: {
|
|
13
17
|
message: "Database record not found",
|
|
14
18
|
statusCode: 404
|
|
@@ -25,6 +29,10 @@ const definitions = {
|
|
|
25
29
|
message: "Database operation temporarily unavailable",
|
|
26
30
|
statusCode: 503
|
|
27
31
|
},
|
|
32
|
+
UNSUPPORTED_MEDIA_TYPE: {
|
|
33
|
+
message: "Database request body must be JSON",
|
|
34
|
+
statusCode: 415
|
|
35
|
+
},
|
|
28
36
|
PAYLOAD_TOO_LARGE: {
|
|
29
37
|
message: "Database response is too large",
|
|
30
38
|
statusCode: 413
|
|
@@ -44,6 +52,8 @@ const categoryByStatus = {
|
|
|
44
52
|
404: "NOT_FOUND",
|
|
45
53
|
409: "CONFLICT",
|
|
46
54
|
413: "PAYLOAD_TOO_LARGE",
|
|
55
|
+
415: "UNSUPPORTED_MEDIA_TYPE",
|
|
56
|
+
422: "VALIDATION_FAILED",
|
|
47
57
|
503: "TRANSIENT"
|
|
48
58
|
};
|
|
49
59
|
/** AppKit-facing database failure with stable metadata and no driver details. */
|
|
@@ -51,9 +61,9 @@ var DatabasePluginError = class extends AppKitError {
|
|
|
51
61
|
code = "DATABASE_PLUGIN_ERROR";
|
|
52
62
|
isRetryable;
|
|
53
63
|
statusCode;
|
|
54
|
-
constructor(category, phase,
|
|
64
|
+
constructor(category, phase, diagnosticMessage, details) {
|
|
55
65
|
const definition = definitions[category];
|
|
56
|
-
super(phase === "runtime" &&
|
|
66
|
+
super((phase === "runtime" || phase === "setup") && diagnosticMessage ? diagnosticMessage : definition.message, { clientMessage: definition.message });
|
|
57
67
|
this.category = category;
|
|
58
68
|
this.phase = phase;
|
|
59
69
|
this.details = details;
|
|
@@ -67,8 +77,8 @@ function invalidDatabaseRequest(runtimeMessage) {
|
|
|
67
77
|
return new DatabasePluginError("INVALID_REQUEST", "runtime", runtimeMessage);
|
|
68
78
|
}
|
|
69
79
|
/** Refuse to publish a plugin whose configuration cannot be honored. */
|
|
70
|
-
function databaseSetupFailed() {
|
|
71
|
-
return new DatabasePluginError("SETUP_FAILED", "setup");
|
|
80
|
+
function databaseSetupFailed(reason) {
|
|
81
|
+
return new DatabasePluginError("SETUP_FAILED", "setup", reason ? `Database setup failed: ${reason}` : void 0);
|
|
72
82
|
}
|
|
73
83
|
/** Reject untrusted request input, naming the field but never its value. */
|
|
74
84
|
function invalidDatabaseInput(path, message) {
|
|
@@ -93,7 +103,7 @@ function describeUnclassifiedError(error) {
|
|
|
93
103
|
}
|
|
94
104
|
/** Add operation context without retaining an unknown error's details. */
|
|
95
105
|
function classifyDatabaseError(error, phase) {
|
|
96
|
-
if (error instanceof DatabasePluginError) return error.phase === phase ? error : new DatabasePluginError(error.category, phase);
|
|
106
|
+
if (error instanceof DatabasePluginError) return error.phase === phase ? error : new DatabasePluginError(error.category, phase, void 0, error.details);
|
|
97
107
|
logger.error("Unclassified database error during %s (%s)", phase, describeUnclassifiedError(error));
|
|
98
108
|
return new DatabasePluginError("INTERNAL", phase);
|
|
99
109
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"errors.js","names":[],"sources":["../../src/database/errors.ts"],"sourcesContent":["import { AppKitError } from \"../errors\";\nimport { createLogger } from \"../logging/logger\";\n\nconst logger = createLogger(\"database\");\n\nexport type DatabaseErrorCategory =\n | \"INVALID_REQUEST\"\n | \"NOT_FOUND\"\n | \"CONFLICT\"\n | \"FORBIDDEN\"\n | \"TRANSIENT\"\n | \"PAYLOAD_TOO_LARGE\"\n | \"INTERNAL\"\n | \"SETUP_FAILED\";\n\n/** Which request field a rejection concerns; it never carries caller values. */\nexport interface DatabaseErrorDetail {\n readonly path: readonly string[];\n readonly message: string;\n}\n\ntype DatabaseErrorPhase =\n | \"setup\"\n | \"shutdown\"\n | \"read\"\n | \"write\"\n | \"transaction\"\n | \"runtime\";\n\nconst definitions: Record<\n DatabaseErrorCategory,\n { readonly message: string; readonly statusCode: number }\n> = {\n INVALID_REQUEST: { message: \"Invalid database request\", statusCode: 400 },\n NOT_FOUND: { message: \"Database record not found\", statusCode: 404 },\n CONFLICT: { message: \"Database conflict\", statusCode: 409 },\n FORBIDDEN: { message: \"Database operation forbidden\", statusCode: 403 },\n TRANSIENT: {\n message: \"Database operation temporarily unavailable\",\n statusCode: 503,\n },\n PAYLOAD_TOO_LARGE: {\n message: \"Database response is too large\",\n statusCode: 413,\n },\n INTERNAL: { message: \"Database operation failed\", statusCode: 500 },\n SETUP_FAILED: { message: \"Database setup failed\", statusCode: 500 },\n};\n\nconst categoryByStatus: Readonly<Record<number, DatabaseErrorCategory>> = {\n 400: \"INVALID_REQUEST\",\n 403: \"FORBIDDEN\",\n 404: \"NOT_FOUND\",\n 409: \"CONFLICT\",\n 413: \"PAYLOAD_TOO_LARGE\",\n 503: \"TRANSIENT\",\n};\n\n/** AppKit-facing database failure with stable metadata and no driver details. */\nexport class DatabasePluginError extends AppKitError {\n readonly code = \"DATABASE_PLUGIN_ERROR\";\n readonly isRetryable: boolean;\n readonly statusCode: number;\n\n constructor(\n readonly category: DatabaseErrorCategory,\n readonly phase: DatabaseErrorPhase,\n
|
|
1
|
+
{"version":3,"file":"errors.js","names":[],"sources":["../../src/database/errors.ts"],"sourcesContent":["import { AppKitError } from \"../errors\";\nimport { createLogger } from \"../logging/logger\";\n\nconst logger = createLogger(\"database\");\n\nexport type DatabaseErrorCategory =\n | \"INVALID_REQUEST\"\n | \"VALIDATION_FAILED\"\n | \"NOT_FOUND\"\n | \"CONFLICT\"\n | \"FORBIDDEN\"\n | \"TRANSIENT\"\n | \"UNSUPPORTED_MEDIA_TYPE\"\n | \"PAYLOAD_TOO_LARGE\"\n | \"INTERNAL\"\n | \"SETUP_FAILED\";\n\n/** Which request field a rejection concerns; it never carries caller values. */\nexport interface DatabaseErrorDetail {\n readonly path: readonly string[];\n readonly message: string;\n}\n\ntype DatabaseErrorPhase =\n | \"setup\"\n | \"shutdown\"\n | \"read\"\n | \"write\"\n | \"transaction\"\n | \"runtime\";\n\nconst definitions: Record<\n DatabaseErrorCategory,\n { readonly message: string; readonly statusCode: number }\n> = {\n INVALID_REQUEST: { message: \"Invalid database request\", statusCode: 400 },\n VALIDATION_FAILED: {\n message: \"Database request failed validation\",\n statusCode: 422,\n },\n NOT_FOUND: { message: \"Database record not found\", statusCode: 404 },\n CONFLICT: { message: \"Database conflict\", statusCode: 409 },\n FORBIDDEN: { message: \"Database operation forbidden\", statusCode: 403 },\n TRANSIENT: {\n message: \"Database operation temporarily unavailable\",\n statusCode: 503,\n },\n UNSUPPORTED_MEDIA_TYPE: {\n message: \"Database request body must be JSON\",\n statusCode: 415,\n },\n PAYLOAD_TOO_LARGE: {\n message: \"Database response is too large\",\n statusCode: 413,\n },\n INTERNAL: { message: \"Database operation failed\", statusCode: 500 },\n SETUP_FAILED: { message: \"Database setup failed\", statusCode: 500 },\n};\n\nconst categoryByStatus: Readonly<Record<number, DatabaseErrorCategory>> = {\n 400: \"INVALID_REQUEST\",\n 403: \"FORBIDDEN\",\n 404: \"NOT_FOUND\",\n 409: \"CONFLICT\",\n 413: \"PAYLOAD_TOO_LARGE\",\n 415: \"UNSUPPORTED_MEDIA_TYPE\",\n 422: \"VALIDATION_FAILED\",\n 503: \"TRANSIENT\",\n};\n\n/** AppKit-facing database failure with stable metadata and no driver details. */\nexport class DatabasePluginError extends AppKitError {\n readonly code = \"DATABASE_PLUGIN_ERROR\";\n readonly isRetryable: boolean;\n readonly statusCode: number;\n\n constructor(\n readonly category: DatabaseErrorCategory,\n readonly phase: DatabaseErrorPhase,\n diagnosticMessage?: string,\n readonly details?: readonly DatabaseErrorDetail[],\n ) {\n const definition = definitions[category];\n // Setup and runtime diagnostics stay server-side. Request boundaries and\n // clientMessage always use the stable message.\n super(\n (phase === \"runtime\" || phase === \"setup\") && diagnosticMessage\n ? diagnosticMessage\n : definition.message,\n {\n clientMessage: definition.message,\n },\n );\n this.statusCode = definition.statusCode;\n this.isRetryable = category === \"TRANSIENT\";\n this.name = \"DatabasePluginError\";\n }\n}\n\n/** Keep runtime diagnostics internal until a plugin boundary classifies them. */\nexport function invalidDatabaseRequest(\n runtimeMessage?: string,\n): DatabasePluginError {\n return new DatabasePluginError(\"INVALID_REQUEST\", \"runtime\", runtimeMessage);\n}\n\n/** Refuse to publish a plugin whose configuration cannot be honored. */\nexport function databaseSetupFailed(reason?: string): DatabasePluginError {\n return new DatabasePluginError(\n \"SETUP_FAILED\",\n \"setup\",\n reason ? `Database setup failed: ${reason}` : undefined,\n );\n}\n\n/** Reject untrusted request input, naming the field but never its value. */\nexport function invalidDatabaseInput(\n path: readonly string[],\n message: string,\n): DatabasePluginError {\n return new DatabasePluginError(\"INVALID_REQUEST\", \"read\", undefined, [\n { path, message },\n ]);\n}\n\n/**\n * Name an unknown failure without its payload. A driver error carries the SQL\n * text and its bound row values (for example a `DrizzleQueryError` retains\n * `query` and `params`), so only constructor names walk into the log.\n */\nfunction describeUnclassifiedError(error: unknown): string {\n const names: string[] = [];\n let current: unknown = error;\n for (let depth = 0; depth < 5 && current instanceof Error; depth++) {\n names.push(current.name);\n current = current.cause;\n }\n return names.length > 0 ? names.join(\" <- \") : typeof error;\n}\n\n/** Add operation context without retaining an unknown error's details. */\nexport function classifyDatabaseError(\n error: unknown,\n phase: DatabaseErrorPhase,\n): DatabasePluginError {\n if (error instanceof DatabasePluginError) {\n // Details name request fields only, so they survive a change of phase.\n return error.phase === phase\n ? error\n : new DatabasePluginError(\n error.category,\n phase,\n undefined,\n error.details,\n );\n }\n logger.error(\n \"Unclassified database error during %s (%s)\",\n phase,\n describeUnclassifiedError(error),\n );\n return new DatabasePluginError(\"INTERNAL\", phase);\n}\n\n/** Restore the safe database category carried through `Plugin.execute()`. */\nexport function databaseErrorFromStatus(\n status: number,\n phase: DatabaseErrorPhase,\n): DatabasePluginError {\n const category = categoryByStatus[status] ?? \"INTERNAL\";\n return new DatabasePluginError(category, phase);\n}\n"],"mappings":";;;;;AAGA,MAAM,SAAS,aAAa,WAAW;AA4BvC,MAAM,cAGF;CACF,iBAAiB;EAAE,SAAS;EAA4B,YAAY;EAAK;CACzE,mBAAmB;EACjB,SAAS;EACT,YAAY;EACb;CACD,WAAW;EAAE,SAAS;EAA6B,YAAY;EAAK;CACpE,UAAU;EAAE,SAAS;EAAqB,YAAY;EAAK;CAC3D,WAAW;EAAE,SAAS;EAAgC,YAAY;EAAK;CACvE,WAAW;EACT,SAAS;EACT,YAAY;EACb;CACD,wBAAwB;EACtB,SAAS;EACT,YAAY;EACb;CACD,mBAAmB;EACjB,SAAS;EACT,YAAY;EACb;CACD,UAAU;EAAE,SAAS;EAA6B,YAAY;EAAK;CACnE,cAAc;EAAE,SAAS;EAAyB,YAAY;EAAK;CACpE;AAED,MAAM,mBAAoE;CACxE,KAAK;CACL,KAAK;CACL,KAAK;CACL,KAAK;CACL,KAAK;CACL,KAAK;CACL,KAAK;CACL,KAAK;CACN;;AAGD,IAAa,sBAAb,cAAyC,YAAY;CACnD,AAAS,OAAO;CAChB,AAAS;CACT,AAAS;CAET,YACE,AAAS,UACT,AAAS,OACT,mBACA,AAAS,SACT;EACA,MAAM,aAAa,YAAY;AAG/B,SACG,UAAU,aAAa,UAAU,YAAY,oBAC1C,oBACA,WAAW,SACf,EACE,eAAe,WAAW,SAC3B,CACF;EAfQ;EACA;EAEA;AAaT,OAAK,aAAa,WAAW;AAC7B,OAAK,cAAc,aAAa;AAChC,OAAK,OAAO;;;;AAKhB,SAAgB,uBACd,gBACqB;AACrB,QAAO,IAAI,oBAAoB,mBAAmB,WAAW,eAAe;;;AAI9E,SAAgB,oBAAoB,QAAsC;AACxE,QAAO,IAAI,oBACT,gBACA,SACA,SAAS,0BAA0B,WAAW,OAC/C;;;AAIH,SAAgB,qBACd,MACA,SACqB;AACrB,QAAO,IAAI,oBAAoB,mBAAmB,QAAQ,QAAW,CACnE;EAAE;EAAM;EAAS,CAClB,CAAC;;;;;;;AAQJ,SAAS,0BAA0B,OAAwB;CACzD,MAAM,QAAkB,EAAE;CAC1B,IAAI,UAAmB;AACvB,MAAK,IAAI,QAAQ,GAAG,QAAQ,KAAK,mBAAmB,OAAO,SAAS;AAClE,QAAM,KAAK,QAAQ,KAAK;AACxB,YAAU,QAAQ;;AAEpB,QAAO,MAAM,SAAS,IAAI,MAAM,KAAK,OAAO,GAAG,OAAO;;;AAIxD,SAAgB,sBACd,OACA,OACqB;AACrB,KAAI,iBAAiB,oBAEnB,QAAO,MAAM,UAAU,QACnB,QACA,IAAI,oBACF,MAAM,UACN,OACA,QACA,MAAM,QACP;AAEP,QAAO,MACL,8CACA,OACA,0BAA0B,MAAM,CACjC;AACD,QAAO,IAAI,oBAAoB,YAAY,MAAM;;;AAInD,SAAgB,wBACd,QACA,OACqB;AAErB,QAAO,IAAI,oBADM,iBAAiB,WAAW,YACJ,MAAM"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"data-path.d.ts","names":[],"sources":["../../../src/database/runtime/data-path.ts"],"mappings":";;;KA8CY,GAAA,GAAM,MAAA"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"data-path.js","names":[],"sources":["../../../src/database/runtime/data-path.ts"],"sourcesContent":["import {\n DEFAULT_LIMIT,\n type FilterOperator,\n type IdValue,\n MAX_LIMIT,\n MAX_OFFSET,\n type OrderDirection,\n} from \"../contract\";\nimport { invalidDatabaseRequest } from \"../errors\";\nimport type { AppKitTable, ColumnMeta } from \"../schema-builder\";\n\nexport type { IdValue, OrderDirection };\nexport type ScalarValue = string | number | bigint | boolean | null;\n/** Operators for one column; array operands are reserved for `in`. */\nexport type FilterOps = Partial<\n Record<FilterOperator, ScalarValue | readonly ScalarValue[]>\n>;\nexport type WhereValue = ScalarValue | readonly ScalarValue[] | FilterOps;\n/** Direct-column predicates with explicit `and` and `or` predicate groups. */\nexport type WhereClause = Readonly<\n Record<string, WhereValue | readonly WhereClause[]>\n>;\n\nexport type OrderSpec = Readonly<Record<string, OrderDirection>>;\n\nexport interface IncludeOptions {\n readonly select?: readonly string[];\n readonly where?: WhereClause;\n readonly order?: OrderSpec;\n readonly limit?: number;\n readonly include?: IncludeSpec;\n}\n\n/** Selection and bounds for one declared relation edge. */\nexport type IncludeSpec = Readonly<Record<string, boolean | IncludeOptions>>;\n\n/** A bounded root read; adapters apply defaults and validate explicit bounds. */\nexport interface QuerySpec {\n readonly where?: WhereClause;\n readonly order?: OrderSpec;\n readonly select?: readonly string[];\n readonly include?: IncludeSpec;\n readonly limit?: number;\n readonly offset?: number;\n}\n\nexport type Row = Record<string, unknown>;\n\n/** Combine predicates without making callers understand the wire shape. */\nexport function andWhere(\n existing: WhereClause | undefined,\n next: WhereClause,\n): WhereClause {\n return existing === undefined ? next : { and: [existing, next] };\n}\n\n/** Internal AppKit execution port; field names are schema-owned identifiers. */\nexport interface DataPath {\n /** Read a bounded collection from one finalized table. */\n select(table: AppKitTable, spec: QuerySpec): Promise<Row[]>;\n /** Read by the sole primary key while preserving supported query state. */\n findOne(\n table: AppKitTable,\n id: IdValue,\n spec?: Pick<QuerySpec, \"where\" | \"select\" | \"include\">,\n ): Promise<Row | null>;\n count(table: AppKitTable, where?: WhereClause): Promise<number>;\n /** Return exactly one inserted row; zero or many is an invariant failure. */\n insert(table: AppKitTable, values: Row): Promise<Row>;\n /** Return null for zero updated rows and reject more than one. */\n update(table: AppKitTable
|
|
1
|
+
{"version":3,"file":"data-path.js","names":[],"sources":["../../../src/database/runtime/data-path.ts"],"sourcesContent":["import {\n DEFAULT_LIMIT,\n type FilterOperator,\n type IdValue,\n MAX_LIMIT,\n MAX_OFFSET,\n type OrderDirection,\n} from \"../contract\";\nimport { invalidDatabaseRequest } from \"../errors\";\nimport type { AppKitTable, ColumnMeta } from \"../schema-builder\";\n\nexport type { IdValue, OrderDirection };\nexport type ScalarValue = string | number | bigint | boolean | null;\n/** Operators for one column; array operands are reserved for `in`. */\nexport type FilterOps = Partial<\n Record<FilterOperator, ScalarValue | readonly ScalarValue[]>\n>;\nexport type WhereValue = ScalarValue | readonly ScalarValue[] | FilterOps;\n/** Direct-column predicates with explicit `and` and `or` predicate groups. */\nexport type WhereClause = Readonly<\n Record<string, WhereValue | readonly WhereClause[]>\n>;\n\nexport type OrderSpec = Readonly<Record<string, OrderDirection>>;\n\nexport interface IncludeOptions {\n readonly select?: readonly string[];\n readonly where?: WhereClause;\n readonly order?: OrderSpec;\n readonly limit?: number;\n readonly include?: IncludeSpec;\n}\n\n/** Selection and bounds for one declared relation edge. */\nexport type IncludeSpec = Readonly<Record<string, boolean | IncludeOptions>>;\n\n/** A bounded root read; adapters apply defaults and validate explicit bounds. */\nexport interface QuerySpec {\n readonly where?: WhereClause;\n readonly order?: OrderSpec;\n readonly select?: readonly string[];\n readonly include?: IncludeSpec;\n readonly limit?: number;\n readonly offset?: number;\n}\n\nexport type Row = Record<string, unknown>;\n\n/** Combine predicates without making callers understand the wire shape. */\nexport function andWhere(\n existing: WhereClause | undefined,\n next: WhereClause,\n): WhereClause {\n return existing === undefined ? next : { and: [existing, next] };\n}\n\n/** Internal AppKit execution port; field names are schema-owned identifiers. */\nexport interface DataPath {\n /** Read a bounded collection from one finalized table. */\n select(table: AppKitTable, spec: QuerySpec): Promise<Row[]>;\n /** Read by the sole primary key while preserving supported query state. */\n findOne(\n table: AppKitTable,\n id: IdValue,\n spec?: Pick<QuerySpec, \"where\" | \"select\" | \"include\">,\n ): Promise<Row | null>;\n count(table: AppKitTable, where?: WhereClause): Promise<number>;\n /** Return exactly one inserted row; zero or many is an invariant failure. */\n insert(table: AppKitTable, values: Row): Promise<Row>;\n /** Return null for zero updated rows and reject more than one. */\n update(\n table: AppKitTable,\n id: IdValue,\n values: Row,\n where?: WhereClause,\n ): Promise<Row | null>;\n /** Return exactly one row for a validated primary-key or unique conflict. */\n upsert(table: AppKitTable, values: Row, onConflict: string): Promise<Row>;\n /** Return false for zero deleted rows, true for one, and reject many. */\n delete(\n table: AppKitTable,\n id: IdValue,\n where?: WhereClause,\n ): Promise<boolean>;\n /** Execute tagged SQL whose interpolations are parameter values, not SQL. */\n raw<T = Row>(\n strings: TemplateStringsArray,\n ...values: unknown[]\n ): Promise<T[]>;\n /** Run the callback with one transaction-bound DataPath. */\n transaction<T>(callback: (tx: DataPath) => Promise<T>): Promise<T>;\n}\n\n/** Validate an explicit root or relation row limit. */\nexport function validateLimit(limit: number): number {\n if (!Number.isInteger(limit) || limit < 0 || limit > MAX_LIMIT) {\n throw invalidDatabaseRequest(\n `limit must be an integer between 0 and ${MAX_LIMIT}`,\n );\n }\n return limit;\n}\n\n/** Apply the conservative collection default when no limit is supplied. */\nexport function limitOrDefault(limit?: number): number {\n return limit === undefined ? DEFAULT_LIMIT : validateLimit(limit);\n}\n\n/** Validate an explicit root or relation row offset. */\nexport function validateOffset(offset: number): number {\n if (!Number.isSafeInteger(offset) || offset < 0 || offset > MAX_OFFSET) {\n throw invalidDatabaseRequest(\n `offset must be an integer between 0 and ${MAX_OFFSET}`,\n );\n }\n return offset;\n}\n\n/** Resolve the sole primary key required by keyed operations. */\nexport function primaryKeyMeta(table: AppKitTable): ColumnMeta {\n const primaryKeys = Object.values(table.$columns).filter(\n (column) => column.primaryKey,\n );\n if (primaryKeys.length !== 1) {\n throw invalidDatabaseRequest(`Table \"${table.$name}\" has no primary key`);\n }\n return primaryKeys[0];\n}\n\n/** Resolve an upsert target that PostgreSQL can use for conflict detection. */\nexport function conflictTargetMeta(\n table: AppKitTable,\n columnName: string,\n): ColumnMeta {\n const column = table.$columns[columnName];\n if (!column || (!column.primaryKey && !column.unique)) {\n throw invalidDatabaseRequest(\n `Column \"${table.$name}.${columnName}\" is not a conflict target`,\n );\n }\n return column;\n}\n"],"mappings":";;;;;;AAiDA,SAAgB,SACd,UACA,MACa;AACb,QAAO,aAAa,SAAY,OAAO,EAAE,KAAK,CAAC,UAAU,KAAK,EAAE;;;AAyClE,SAAgB,cAAc,OAAuB;AACnD,KAAI,CAAC,OAAO,UAAU,MAAM,IAAI,QAAQ,KAAK,QAAQ,UACnD,OAAM,uBACJ,0CAA0C,YAC3C;AAEH,QAAO;;;AAIT,SAAgB,eAAe,OAAwB;AACrD,QAAO,UAAU,SAAY,gBAAgB,cAAc,MAAM;;;AAInE,SAAgB,eAAe,QAAwB;AACrD,KAAI,CAAC,OAAO,cAAc,OAAO,IAAI,SAAS,KAAK,SAAS,WAC1D,OAAM,uBACJ,2CAA2C,aAC5C;AAEH,QAAO;;;AAIT,SAAgB,eAAe,OAAgC;CAC7D,MAAM,cAAc,OAAO,OAAO,MAAM,SAAS,CAAC,QAC/C,WAAW,OAAO,WACpB;AACD,KAAI,YAAY,WAAW,EACzB,OAAM,uBAAuB,UAAU,MAAM,MAAM,sBAAsB;AAE3E,QAAO,YAAY;;;AAIrB,SAAgB,mBACd,OACA,YACY;CACZ,MAAM,SAAS,MAAM,SAAS;AAC9B,KAAI,CAAC,UAAW,CAAC,OAAO,cAAc,CAAC,OAAO,OAC5C,OAAM,uBACJ,WAAW,MAAM,MAAM,GAAG,WAAW,4BACtC;AAEH,QAAO"}
|
|
@@ -2,7 +2,7 @@ import { createLogger } from "../../../logging/logger.js";
|
|
|
2
2
|
import { columnValueSchema } from "../../schema-builder/validators.js";
|
|
3
3
|
import { DatabasePluginError, invalidDatabaseRequest } from "../../errors.js";
|
|
4
4
|
import { buildEngineRelations } from "../../schema-builder/engine/relations.js";
|
|
5
|
-
import { conflictTargetMeta, limitOrDefault, primaryKeyMeta, validateOffset } from "../data-path.js";
|
|
5
|
+
import { andWhere, conflictTargetMeta, limitOrDefault, primaryKeyMeta, validateOffset } from "../data-path.js";
|
|
6
6
|
import { columnOf, defaultColumns, publicColumnNames, returningColumns, selectToColumns, translateInclude, translateOrder, translateWhere } from "./translate.js";
|
|
7
7
|
import { and, eq, isSQLWrapper, sql } from "drizzle-orm";
|
|
8
8
|
import { drizzle } from "drizzle-orm/node-postgres";
|
|
@@ -113,6 +113,8 @@ function createDrizzleDataPath(db, schema, options = {}) {
|
|
|
113
113
|
return table.$engine;
|
|
114
114
|
};
|
|
115
115
|
const whereSql = (table, where) => where === void 0 ? void 0 : translateWhere(table, where, columnAccess);
|
|
116
|
+
/** Narrow by the primary key while preserving an accumulated predicate. */
|
|
117
|
+
const keyedWhere = (table, primaryKey, id, where) => where === void 0 ? eq(columnOf(table, primaryKey.columnName), id) : translateWhere(table, andWhere({ [primaryKey.columnName]: { eq: id } }, where), columnAccess);
|
|
116
118
|
return {
|
|
117
119
|
async select(table, spec) {
|
|
118
120
|
return runDatabaseOperation(() => relationalQueryBuilder(db, schema, table).findMany({
|
|
@@ -142,11 +144,11 @@ function createDrizzleDataPath(db, schema, options = {}) {
|
|
|
142
144
|
const parameters = mutationValues(table, values, "insert");
|
|
143
145
|
return expectExactlyOne(mutationRows(table, await runDatabaseOperation(() => db.insert(engineTable).values(parameters).returning(returningColumns(table, columnAccess))), columnAccess));
|
|
144
146
|
},
|
|
145
|
-
async update(table, id, values) {
|
|
147
|
+
async update(table, id, values, where) {
|
|
146
148
|
const engineTable = pgTable(table);
|
|
147
149
|
const parameters = mutationValues(table, values, "update");
|
|
148
150
|
const { meta: primaryKey, value: validatedId } = validatedPrimaryKey(table, id);
|
|
149
|
-
return expectZeroOrOne(mutationRows(table, await runDatabaseOperation(() => db.update(engineTable).set(parameters).where(
|
|
151
|
+
return expectZeroOrOne(mutationRows(table, await runDatabaseOperation(() => db.update(engineTable).set(parameters).where(keyedWhere(table, primaryKey, validatedId, where)).returning(returningColumns(table, columnAccess))), columnAccess));
|
|
150
152
|
},
|
|
151
153
|
async upsert(table, values, onConflict) {
|
|
152
154
|
const engineTable = pgTable(table);
|
|
@@ -158,9 +160,9 @@ function createDrizzleDataPath(db, schema, options = {}) {
|
|
|
158
160
|
set: updates
|
|
159
161
|
}).returning(returningColumns(table, columnAccess))), columnAccess));
|
|
160
162
|
},
|
|
161
|
-
async delete(table, id) {
|
|
163
|
+
async delete(table, id, where) {
|
|
162
164
|
const { meta: primaryKey, value: validatedId } = validatedPrimaryKey(table, id);
|
|
163
|
-
return expectZeroOrOne(await runDatabaseOperation(() => db.delete(pgTable(table)).where(
|
|
165
|
+
return expectZeroOrOne(await runDatabaseOperation(() => db.delete(pgTable(table)).where(keyedWhere(table, primaryKey, validatedId, where)).returning({ id: columnOf(table, primaryKey.columnName) }))) !== null;
|
|
164
166
|
},
|
|
165
167
|
async raw(strings, ...values) {
|
|
166
168
|
if (values.some((value) => isSQLWrapper(value))) throw invalidDatabaseRequest("Tagged SQL interpolations must be parameter values");
|