@databricks/appkit 0.70.0 → 0.72.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/CLAUDE.md +46 -0
  2. package/NOTICE.md +2 -2
  3. package/dist/appkit/package.js +1 -1
  4. package/dist/beta.d.ts +13 -1
  5. package/dist/beta.js +13 -1
  6. package/dist/cli/commands/agent/eval.js +112 -0
  7. package/dist/cli/commands/agent/eval.js.map +1 -0
  8. package/dist/cli/commands/agent/index.js +18 -0
  9. package/dist/cli/commands/agent/index.js.map +1 -0
  10. package/dist/cli/index.js +2 -0
  11. package/dist/cli/index.js.map +1 -1
  12. package/dist/connectors/index.js +2 -0
  13. package/dist/connectors/mlflow/auth.d.ts +18 -0
  14. package/dist/connectors/mlflow/auth.d.ts.map +1 -0
  15. package/dist/connectors/mlflow/auth.js +50 -0
  16. package/dist/connectors/mlflow/auth.js.map +1 -0
  17. package/dist/connectors/mlflow/client.d.ts +51 -0
  18. package/dist/connectors/mlflow/client.d.ts.map +1 -0
  19. package/dist/connectors/mlflow/client.js +93 -0
  20. package/dist/connectors/mlflow/client.js.map +1 -0
  21. package/dist/core/appkit.d.ts.map +1 -1
  22. package/dist/core/appkit.js +1 -0
  23. package/dist/core/appkit.js.map +1 -1
  24. package/dist/evals/define-eval.d.ts +26 -0
  25. package/dist/evals/define-eval.d.ts.map +1 -0
  26. package/dist/evals/define-eval.js +28 -0
  27. package/dist/evals/define-eval.js.map +1 -0
  28. package/dist/evals/discover.d.ts +20 -0
  29. package/dist/evals/discover.d.ts.map +1 -0
  30. package/dist/evals/discover.js +49 -0
  31. package/dist/evals/discover.js.map +1 -0
  32. package/dist/evals/http-driver.d.ts +33 -0
  33. package/dist/evals/http-driver.d.ts.map +1 -0
  34. package/dist/evals/http-driver.js +118 -0
  35. package/dist/evals/http-driver.js.map +1 -0
  36. package/dist/evals/index.js +13 -0
  37. package/dist/evals/judge.d.ts +26 -0
  38. package/dist/evals/judge.d.ts.map +1 -0
  39. package/dist/evals/judge.js +77 -0
  40. package/dist/evals/judge.js.map +1 -0
  41. package/dist/evals/matchers.d.ts +12 -0
  42. package/dist/evals/matchers.d.ts.map +1 -0
  43. package/dist/evals/matchers.js +26 -0
  44. package/dist/evals/matchers.js.map +1 -0
  45. package/dist/evals/mlflow-report.d.ts +36 -0
  46. package/dist/evals/mlflow-report.d.ts.map +1 -0
  47. package/dist/evals/mlflow-report.js +161 -0
  48. package/dist/evals/mlflow-report.js.map +1 -0
  49. package/dist/evals/mlflow-run.d.ts +11 -0
  50. package/dist/evals/mlflow-run.d.ts.map +1 -0
  51. package/dist/evals/mlflow-run.js +101 -0
  52. package/dist/evals/mlflow-run.js.map +1 -0
  53. package/dist/evals/pool.js +24 -0
  54. package/dist/evals/pool.js.map +1 -0
  55. package/dist/evals/report.d.ts +25 -0
  56. package/dist/evals/report.d.ts.map +1 -0
  57. package/dist/evals/report.js +57 -0
  58. package/dist/evals/report.js.map +1 -0
  59. package/dist/evals/run-eval.d.ts +20 -0
  60. package/dist/evals/run-eval.d.ts.map +1 -0
  61. package/dist/evals/run-eval.js +144 -0
  62. package/dist/evals/run-eval.js.map +1 -0
  63. package/dist/evals/run-evals.d.ts +85 -0
  64. package/dist/evals/run-evals.d.ts.map +1 -0
  65. package/dist/evals/run-evals.js +178 -0
  66. package/dist/evals/run-evals.js.map +1 -0
  67. package/dist/evals/types.d.ts +127 -0
  68. package/dist/evals/types.d.ts.map +1 -0
  69. package/dist/plugins/agents/agents.d.ts.map +1 -1
  70. package/dist/plugins/agents/agents.js +5 -2
  71. package/dist/plugins/agents/agents.js.map +1 -1
  72. package/dist/plugins/agents/mlflow.js +267 -11
  73. package/dist/plugins/agents/mlflow.js.map +1 -1
  74. package/dist/plugins/server/index.js +2 -2
  75. package/dist/plugins/server/index.js.map +1 -1
  76. package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js +3 -3
  77. package/dist/plugins/server/remote-tunnel/remote-tunnel-manager.js.map +1 -1
  78. package/dist/plugins/server/static-server.js +3 -3
  79. package/dist/plugins/server/static-server.js.map +1 -1
  80. package/dist/plugins/server/utils.js +3 -3
  81. package/dist/plugins/server/utils.js.map +1 -1
  82. package/dist/plugins/server/vite-dev-server.js +4 -4
  83. package/dist/plugins/server/vite-dev-server.js.map +1 -1
  84. package/dist/shared/src/schemas/manifest.d.ts +33 -33
  85. package/dist/telemetry/index.js +2 -2
  86. package/dist/telemetry/telemetry-manager.d.ts +2 -1
  87. package/dist/telemetry/telemetry-manager.js +114 -26
  88. package/dist/telemetry/telemetry-manager.js.map +1 -1
  89. package/dist/type-generator/database/generate.js +3 -3
  90. package/dist/type-generator/database/generate.js.map +1 -1
  91. package/dist/type-generator/errors.js +45 -11
  92. package/dist/type-generator/errors.js.map +1 -1
  93. package/dist/type-generator/migration.js +2 -2
  94. package/dist/type-generator/migration.js.map +1 -1
  95. package/dist/type-generator/mv-registry/sync.js +3 -3
  96. package/dist/type-generator/mv-registry/sync.js.map +1 -1
  97. package/dist/type-generator/mv-registry/types.d.ts +5 -2
  98. package/dist/type-generator/mv-registry/types.d.ts.map +1 -1
  99. package/dist/type-generator/query-registry.js +3 -3
  100. package/dist/type-generator/query-registry.js.map +1 -1
  101. package/dist/type-generator/serving/server-file-extractor.js +3 -3
  102. package/dist/type-generator/serving/server-file-extractor.js.map +1 -1
  103. package/docs/api/appkit/Class.MlflowClient.md +103 -0
  104. package/docs/api/appkit/Function.buildAssessments.md +16 -0
  105. package/docs/api/appkit/Function.configureJudge.md +18 -0
  106. package/docs/api/appkit/Function.createHttpDriver.md +18 -0
  107. package/docs/api/appkit/Function.defineEval.md +35 -0
  108. package/docs/api/appkit/Function.discoverEvalFiles.md +18 -0
  109. package/docs/api/appkit/Function.equals.md +18 -0
  110. package/docs/api/appkit/Function.evalGlyph.md +18 -0
  111. package/docs/api/appkit/Function.formatEvalDetail.md +18 -0
  112. package/docs/api/appkit/Function.formatEvalHeadline.md +18 -0
  113. package/docs/api/appkit/Function.formatEvalResults.md +18 -0
  114. package/docs/api/appkit/Function.formatSummaryLine.md +18 -0
  115. package/docs/api/appkit/Function.includes.md +18 -0
  116. package/docs/api/appkit/Function.isJudgeConfigured.md +10 -0
  117. package/docs/api/appkit/Function.matches.md +18 -0
  118. package/docs/api/appkit/Function.normalizeHost.md +18 -0
  119. package/docs/api/appkit/Function.reportToMlflow.md +23 -0
  120. package/docs/api/appkit/Function.resolveDatabricksAuth.md +16 -0
  121. package/docs/api/appkit/Function.runEval.md +19 -0
  122. package/docs/api/appkit/Function.runEvalsInDir.md +18 -0
  123. package/docs/api/appkit/Function.summarize.md +16 -0
  124. package/docs/api/appkit/Interface.AssertionHandle.md +54 -0
  125. package/docs/api/appkit/Interface.AssertionResult.md +48 -0
  126. package/docs/api/appkit/Interface.Assessment.md +83 -0
  127. package/docs/api/appkit/Interface.CustomJudgeSpec.md +30 -0
  128. package/docs/api/appkit/Interface.DatabricksAuth.md +21 -0
  129. package/docs/api/appkit/Interface.DiscoveredEval.md +36 -0
  130. package/docs/api/appkit/Interface.DriveResult.md +58 -0
  131. package/docs/api/appkit/Interface.EvalDefinition.md +46 -0
  132. package/docs/api/appkit/Interface.EvalDriver.md +22 -0
  133. package/docs/api/appkit/Interface.EvalResult.md +83 -0
  134. package/docs/api/appkit/Interface.EvalRunSummary.md +46 -0
  135. package/docs/api/appkit/Interface.EvalSummary.md +48 -0
  136. package/docs/api/appkit/Interface.HttpDriverOptions.md +67 -0
  137. package/docs/api/appkit/Interface.JudgeConfig.md +34 -0
  138. package/docs/api/appkit/Interface.JudgeScore.md +21 -0
  139. package/docs/api/appkit/Interface.MatchResult.md +34 -0
  140. package/docs/api/appkit/Interface.PostResult.md +30 -0
  141. package/docs/api/appkit/Interface.ReportOutcome.md +53 -0
  142. package/docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md +34 -0
  143. package/docs/api/appkit/Interface.RunEvalOptions.md +34 -0
  144. package/docs/api/appkit/Interface.RunEvalsOptions.md +192 -0
  145. package/docs/api/appkit/Interface.TestContext.md +208 -0
  146. package/docs/api/appkit/TypeAlias.EvalProgress.md +26 -0
  147. package/docs/api/appkit/TypeAlias.Matcher.md +18 -0
  148. package/docs/api/appkit/TypeAlias.Severity.md +8 -0
  149. package/docs/api/appkit.md +63 -17
  150. package/llms.txt +46 -0
  151. package/package.json +5 -4
  152. package/sbom.cdx.json +1 -1
package/CLAUDE.md CHANGED
@@ -71,6 +71,7 @@ npx @databricks/appkit docs <query>
71
71
  - [Class: DatabricksAdapter](./docs/api/appkit/Class.DatabricksAdapter.md): Adapter that talks directly to Databricks Model Serving /invocations endpoint.
72
72
  - [Class: ExecutionError](./docs/api/appkit/Class.ExecutionError.md): Error thrown when an operation execution fails.
73
73
  - [Class: InitializationError](./docs/api/appkit/Class.InitializationError.md): Error thrown when a service or component is not properly initialized.
74
+ - [Class: MlflowClient](./docs/api/appkit/Class.MlflowClient.md): A thin client over the Databricks workspace REST API, owning the host + bearer
74
75
  - [Abstract Class: Plugin<TConfig>](./docs/api/appkit/Class.Plugin.md): Base abstract class for creating AppKit plugins.
75
76
  - [Class: PolicyDeniedError](./docs/api/appkit/Class.PolicyDeniedError.md): Thrown when a policy denies an action.
76
77
  - [Class: ResourceRegistry](./docs/api/appkit/Class.ResourceRegistry.md): Central registry for tracking plugin resource requirements.
@@ -86,20 +87,31 @@ npx @databricks/appkit docs <query>
86
87
  - [Function: bigid()](./docs/api/appkit/Function.bigid.md): Returns
87
88
  - [Function: bigint()](./docs/api/appkit/Function.bigint.md): Returns
88
89
  - [Function: boolean()](./docs/api/appkit/Function.boolean.md): Returns
90
+ - [Function: buildAssessments()](./docs/api/appkit/Function.buildAssessments.md): Parameters
91
+ - [Function: configureJudge()](./docs/api/appkit/Function.configureJudge.md): Configure the judge once. Sets the OpenAI-compatible client env autoevals
89
92
  - [Function: createAgent()](./docs/api/appkit/Function.createAgent.md): Pure factory for agent definitions: cycle-detects the sub-agent graph and
90
93
  - [Function: createApp()](./docs/api/appkit/Function.createApp.md): Bootstraps AppKit with the provided configuration.
94
+ - [Function: createHttpDriver()](./docs/api/appkit/Function.createHttpDriver.md): Drives an agent by POSTing to a running app's chat endpoint and parsing the
91
95
  - [Function: createLakebasePool()](./docs/api/appkit/Function.createLakebasePool.md): Create a Lakebase pool with appkit's logger integration.
92
96
  - [Function: createLakebasePoolManager()](./docs/api/appkit/Function.createLakebasePoolManager.md): Create a pool manager that maintains per-key Lakebase connection pools.
93
97
  - [Function: createWorkspaceClient()](./docs/api/appkit/Function.createWorkspaceClient.md): Construct an AppKit workspace client.
94
98
  - [Function: database()](./docs/api/appkit/Function.database.md): Create a typed database plugin registration for a finalized schema.
99
+ - [Function: defineEval()](./docs/api/appkit/Function.defineEval.md): Define an agent eval. Default-export the result from a
95
100
  - [Function: defineManifest()](./docs/api/appkit/Function.defineManifest.md): Validates a raw manifest (typically a manifest.json import) against the
96
101
  - [Function: defineSchema()](./docs/api/appkit/Function.defineSchema.md): Compile one declared schema. The returned type keeps the table names the
97
102
  - [Function: defineTool()](./docs/api/appkit/Function.defineTool.md): Defines a single tool entry for a plugin's internal registry.
103
+ - [Function: discoverEvalFiles()](./docs/api/appkit/Function.discoverEvalFiles.md): Discover evals under /server/agents//evals/ — co-located
98
104
  - [Function: enumColumn()](./docs/api/appkit/Function.enumColumn.md): Parameters
105
+ - [Function: equals()](./docs/api/appkit/Function.equals.md): Passes when the value equals expected exactly.
106
+ - [Function: evalGlyph()](./docs/api/appkit/Function.evalGlyph.md): Status glyph for a single eval result.
99
107
  - [Function: executeFromRegistry()](./docs/api/appkit/Function.executeFromRegistry.md): Validates tool-call arguments against the entry's schema and invokes its
100
108
  - [Function: extractServingEndpoints()](./docs/api/appkit/Function.extractServingEndpoints.md): Extract serving endpoint config from a server file by AST-parsing it.
101
109
  - [Function: findServerFile()](./docs/api/appkit/Function.findServerFile.md): Find the server entry file by checking candidate paths in order.
102
110
  - [Function: fk()](./docs/api/appkit/Function.fk.md): Declare foreign-key to another column.
111
+ - [Function: formatEvalDetail()](./docs/api/appkit/Function.formatEvalDetail.md): Indented detail lines for a failing eval (error + failing assertions).
112
+ - [Function: formatEvalHeadline()](./docs/api/appkit/Function.formatEvalHeadline.md): The one-line header for a single eval result (no failure detail).
113
+ - [Function: formatEvalResults()](./docs/api/appkit/Function.formatEvalResults.md): Render all results as a human-readable console report (non-streaming).
114
+ - [Function: formatSummaryLine()](./docs/api/appkit/Function.formatSummaryLine.md): The final PASS/FAIL summary line.
103
115
  - [Function: fromSupervisorApi()](./docs/api/appkit/Function.fromSupervisorApi.md): Creates an AgentAdapter backed by the Databricks AI Gateway
104
116
  - [Function: functionToolToDefinition()](./docs/api/appkit/Function.functionToolToDefinition.md): Parameters
105
117
  - [Function: generateDatabaseCredential()](./docs/api/appkit/Function.generateDatabaseCredential.md): Generate OAuth credentials for Postgres database connection using the proper Postgres API.
@@ -111,19 +123,28 @@ npx @databricks/appkit docs <query>
111
123
  - [Function: getUsernameWithApiLookup()](./docs/api/appkit/Function.getUsernameWithApiLookup.md): Resolves the PostgreSQL username for a Lakebase connection.
112
124
  - [Function: getWorkspaceClient()](./docs/api/appkit/Function.getWorkspaceClient.md): Get workspace client from config or SDK default auth chain
113
125
  - [Function: id()](./docs/api/appkit/Function.id.md): Returns
126
+ - [Function: includes()](./docs/api/appkit/Function.includes.md): Passes when the value contains substring.
114
127
  - [Function: integer()](./docs/api/appkit/Function.integer.md): Returns
115
128
  - [Function: isFunctionTool()](./docs/api/appkit/Function.isFunctionTool.md): Parameters
116
129
  - [Function: isHostedTool()](./docs/api/appkit/Function.isHostedTool.md): Parameters
130
+ - [Function: isJudgeConfigured()](./docs/api/appkit/Function.isJudgeConfigured.md): Returns
117
131
  - [Function: isSQLTypeMarker()](./docs/api/appkit/Function.isSQLTypeMarker.md): Type guard to check if a value is a SQL type marker
118
132
  - [Function: isSupervisorTool()](./docs/api/appkit/Function.isSupervisorTool.md): Type guard for HostedSupervisorTool. Used by the agents plugin
119
133
  - [Function: isToolkitEntry()](./docs/api/appkit/Function.isToolkitEntry.md): Type guard for ToolkitEntry — used by the agents plugin to differentiate
120
134
  - [Function: jsonb()](./docs/api/appkit/Function.jsonb.md): Returns
121
135
  - [Function: loadAgentFromFile()](./docs/api/appkit/Function.loadAgentFromFile.md): Loads a single markdown agent file and resolves its frontmatter against
122
136
  - [Function: loadAgentsFromDir()](./docs/api/appkit/Function.loadAgentsFromDir.md): Scans a directory for one subdirectory per agent, each containing
137
+ - [Function: matches()](./docs/api/appkit/Function.matches.md): Passes when the value matches pattern.
123
138
  - [Function: mcpServer()](./docs/api/appkit/Function.mcpServer.md): Factory for declaring a custom MCP server tool.
139
+ - [Function: normalizeHost()](./docs/api/appkit/Function.normalizeHost.md): Ensure the host has a scheme (Databricks env often lacks https://).
124
140
  - [Function: parseTextToolCalls()](./docs/api/appkit/Function.parseTextToolCalls.md): Parses text-based tool calls from model output.
141
+ - [Function: reportToMlflow()](./docs/api/appkit/Function.reportToMlflow.md): Write one pass/fail assessment per eval result to the Databricks MLflow REST
142
+ - [Function: resolveDatabricksAuth()](./docs/api/appkit/Function.resolveDatabricksAuth.md): Parameters
125
143
  - [Function: resolveHostedTools()](./docs/api/appkit/Function.resolveHostedTools.md): Parameters
126
144
  - [Function: runAgent()](./docs/api/appkit/Function.runAgent.md): Standalone agent execution without createApp. Resolves the adapter, binds
145
+ - [Function: runEval()](./docs/api/appkit/Function.runEval.md): Run a single eval against a driver. Never throws for assertion or agent
146
+ - [Function: runEvalsInDir()](./docs/api/appkit/Function.runEvalsInDir.md): Discover, load, and run every eval under each agent's evals/ dir, driving
147
+ - [Function: summarize()](./docs/api/appkit/Function.summarize.md): Parameters
127
148
  - [Function: text()](./docs/api/appkit/Function.text.md): Returns
128
149
  - [Function: timestamp()](./docs/api/appkit/Function.timestamp.md): Parameters
129
150
  - [Function: tool()](./docs/api/appkit/Function.tool.md): Factory for defining function tools with Zod schemas.
@@ -136,18 +157,31 @@ npx @databricks/appkit docs <query>
136
157
  - [Interface: AgentRunContext](./docs/api/appkit/Interface.AgentRunContext.md): Properties
137
158
  - [Interface: AgentsPluginConfig](./docs/api/appkit/Interface.AgentsPluginConfig.md): Base configuration interface for AppKit plugins
138
159
  - [Interface: AgentToolDefinition](./docs/api/appkit/Interface.AgentToolDefinition.md): Properties
160
+ - [Interface: AssertionHandle](./docs/api/appkit/Interface.AssertionHandle.md): Chainable handle returned by every assertion to control its severity.
161
+ - [Interface: AssertionResult](./docs/api/appkit/Interface.AssertionResult.md): A single recorded assertion outcome.
162
+ - [Interface: Assessment](./docs/api/appkit/Interface.Assessment.md): A Feedback assessment in the MLflow REST proto-JSON shape.
139
163
  - [Interface: AutoInheritToolsConfig](./docs/api/appkit/Interface.AutoInheritToolsConfig.md): Auto-inherit configuration. When enabled for a given agent origin, agents
140
164
  - [Interface: BasePluginConfig](./docs/api/appkit/Interface.BasePluginConfig.md): Base configuration interface for AppKit plugins
141
165
  - [Interface: CacheConfig](./docs/api/appkit/Interface.CacheConfig.md): Configuration for the CacheInterceptor. Controls TTL, size limits, storage backend, and probabilistic cleanup.
166
+ - [Interface: CustomJudgeSpec](./docs/api/appkit/Interface.CustomJudgeSpec.md): A custom LLM-judge definition: a prompt template and choice→score mapping.
142
167
  - [Interface: DatabaseCredential](./docs/api/appkit/Interface.DatabaseCredential.md): Database credentials with OAuth token for Postgres connection
143
168
  - [Interface: DatabaseRegistry](./docs/api/appkit/Interface.DatabaseRegistry.md): CANONICAL augmentation target. Empty by default; the generated database.d.ts
169
+ - [Interface: DatabricksAuth](./docs/api/appkit/Interface.DatabricksAuth.md): Resolved Databricks host + bearer token for the eval runner's REST calls.
170
+ - [Interface: DiscoveredEval](./docs/api/appkit/Interface.DiscoveredEval.md): An eval file found under server/agents//evals/.
171
+ - [Interface: DriveResult](./docs/api/appkit/Interface.DriveResult.md): What a driver returns for a single t.send.
144
172
  - [Interface: EndpointConfig](./docs/api/appkit/Interface.EndpointConfig.md): Properties
173
+ - [Interface: EvalDefinition](./docs/api/appkit/Interface.EvalDefinition.md): A single eval, default-exported from a *.eval.ts file.
174
+ - [Interface: EvalDriver](./docs/api/appkit/Interface.EvalDriver.md): Abstraction over how the agent is driven. The HTTP driver posts to a running
175
+ - [Interface: EvalResult](./docs/api/appkit/Interface.EvalResult.md): The outcome of running one eval.
176
+ - [Interface: EvalRunSummary](./docs/api/appkit/Interface.EvalRunSummary.md): Properties
177
+ - [Interface: EvalSummary](./docs/api/appkit/Interface.EvalSummary.md): Properties
145
178
  - [Interface: FilePolicyUser](./docs/api/appkit/Interface.FilePolicyUser.md): Minimal user identity passed to the policy function.
146
179
  - [Interface: FileResource](./docs/api/appkit/Interface.FileResource.md): Describes the file or directory being acted upon.
147
180
  - [Interface: FunctionTool](./docs/api/appkit/Interface.FunctionTool.md): Properties
148
181
  - [Interface: GenerateDatabaseCredentialRequest](./docs/api/appkit/Interface.GenerateDatabaseCredentialRequest.md): Request parameters for generating database OAuth credentials
149
182
  - [Interface: GenerationParams](./docs/api/appkit/Interface.GenerationParams.md): Optional generation parameters forwarded to the OpenAI-compatible serving
150
183
  - [Interface: HostedSupervisorTool](./docs/api/appkit/Interface.HostedSupervisorTool.md): Tagged record returned by every supervisorTools factory. The
184
+ - [Interface: HttpDriverOptions](./docs/api/appkit/Interface.HttpDriverOptions.md): Properties
151
185
  - [Interface: IAiSearchConfig](./docs/api/appkit/Interface.IAiSearchConfig.md): Base configuration interface for AppKit plugins
152
186
  - [Interface: IJobsConfig](./docs/api/appkit/Interface.IJobsConfig.md): Configuration for the Jobs plugin.
153
187
  - [Interface: IndexConfig](./docs/api/appkit/Interface.IndexConfig.md): Properties
@@ -155,22 +189,30 @@ npx @databricks/appkit docs <query>
155
189
  - [Interface: JobAPI](./docs/api/appkit/Interface.JobAPI.md): User-facing API for a single configured job.
156
190
  - [Interface: JobConfig](./docs/api/appkit/Interface.JobConfig.md): Per-job configuration options.
157
191
  - [Interface: JobsConnectorConfig](./docs/api/appkit/Interface.JobsConnectorConfig.md): Properties
192
+ - [Interface: JudgeConfig](./docs/api/appkit/Interface.JudgeConfig.md): Properties
193
+ - [Interface: JudgeScore](./docs/api/appkit/Interface.JudgeScore.md): A normalized judge result. score is 0..1.
158
194
  - [Interface: LakebasePool](./docs/api/appkit/Interface.LakebasePool.md): Subset of pg.Pool exposed by the Lakebase plugin.
159
195
  - [Interface: LakebasePoolConfig](./docs/api/appkit/Interface.LakebasePoolConfig.md): Configuration for creating a Lakebase connection pool
160
196
  - [Interface: LakebasePoolManager](./docs/api/appkit/Interface.LakebasePoolManager.md): Manages multiple Lakebase connection pools keyed by an identifier (e.g. userId).
197
+ - [Interface: MatchResult](./docs/api/appkit/Interface.MatchResult.md): Result of a deterministic matcher run against a value.
161
198
  - [Interface: McpConnectAllResult](./docs/api/appkit/Interface.McpConnectAllResult.md): Per-endpoint outcome of AppKitMcpClient.connectAll. Callers (the
162
199
  - [Interface: Message](./docs/api/appkit/Interface.Message.md): Properties
163
200
  - [Interface: PluginManifest<TName>](./docs/api/appkit/Interface.PluginManifest.md): Plugin manifest that declares metadata and resource requirements.
164
201
  - [Interface: PluginToolkitProvider](./docs/api/appkit/Interface.PluginToolkitProvider.md): Minimum shape every entry in the Plugins map must expose. Core
202
+ - [Interface: PostResult](./docs/api/appkit/Interface.PostResult.md): Structured result for a best-effort POST that must not throw.
165
203
  - [Interface: PromptContext](./docs/api/appkit/Interface.PromptContext.md): Context passed to baseSystemPrompt callbacks.
166
204
  - [Interface: RegisteredAgent](./docs/api/appkit/Interface.RegisteredAgent.md): Properties
205
+ - [Interface: ReportOutcome](./docs/api/appkit/Interface.ReportOutcome.md): Properties
167
206
  - [Interface: RequestedClaims](./docs/api/appkit/Interface.RequestedClaims.md): Optional claims for fine-grained Unity Catalog table permissions
168
207
  - [Interface: RequestedResource](./docs/api/appkit/Interface.RequestedResource.md): Resource to request permissions for in Unity Catalog
169
208
  - [Interface: RerankerConfig](./docs/api/appkit/Interface.RerankerConfig.md): Properties
209
+ - [Interface: ResolveDatabricksAuthOptions](./docs/api/appkit/Interface.ResolveDatabricksAuthOptions.md): Properties
170
210
  - [Interface: ResourceEntry](./docs/api/appkit/Interface.ResourceEntry.md): Internal representation of a resource in the registry.
171
211
  - [Interface: ResourceRequirement](./docs/api/appkit/Interface.ResourceRequirement.md): Declares a resource requirement for a plugin.
172
212
  - [Interface: RunAgentInput](./docs/api/appkit/Interface.RunAgentInput.md): Properties
173
213
  - [Interface: RunAgentResult](./docs/api/appkit/Interface.RunAgentResult.md): Properties
214
+ - [Interface: RunEvalOptions](./docs/api/appkit/Interface.RunEvalOptions.md): Properties
215
+ - [Interface: RunEvalsOptions](./docs/api/appkit/Interface.RunEvalsOptions.md): Properties
174
216
  - [Interface: Schema<TTableName>](./docs/api/appkit/Interface.Schema.md): One finalized schema. TTableName keeps the declared names in the type, so
175
217
  - [Interface: SearchRequest](./docs/api/appkit/Interface.SearchRequest.md): Properties
176
218
  - [Interface: SearchResponse<T>](./docs/api/appkit/Interface.SearchResponse.md): Type Parameters
@@ -181,6 +223,7 @@ npx @databricks/appkit docs <query>
181
223
  - [Interface: SupervisorApiAdapterOptions](./docs/api/appkit/Interface.SupervisorApiAdapterOptions.md): Properties
182
224
  - [Interface: SupervisorExtension](./docs/api/appkit/Interface.SupervisorExtension.md): Shape of the value at AgentInput.extensions[SUPERVISOREXTENSIONKEY].
183
225
  - [Interface: TelemetryConfig](./docs/api/appkit/Interface.TelemetryConfig.md): OpenTelemetry configuration for AppKit applications
226
+ - [Interface: TestContext](./docs/api/appkit/Interface.TestContext.md): The t context passed to an eval's test function.
184
227
  - [Interface: Thread](./docs/api/appkit/Interface.Thread.md): Properties
185
228
  - [Interface: ThreadStore](./docs/api/appkit/Interface.ThreadStore.md): Methods
186
229
  - [Interface: ToolAnnotations](./docs/api/appkit/Interface.ToolAnnotations.md): Properties
@@ -200,6 +243,7 @@ npx @databricks/appkit docs <query>
200
243
  - [Type Alias: BaseSystemPromptOption](./docs/api/appkit/TypeAlias.BaseSystemPromptOption.md)
201
244
  - [Type Alias: ConfigSchema](./docs/api/appkit/TypeAlias.ConfigSchema.md): Configuration schema definition for plugin config.
202
245
  - [Type Alias: DatabaseExports](./docs/api/appkit/TypeAlias.DatabaseExports.md): Typed database API published by the plugin.
246
+ - [Type Alias: EvalProgress](./docs/api/appkit/TypeAlias.EvalProgress.md)
203
247
  - [Type Alias: ExecutionResult<T>](./docs/api/appkit/TypeAlias.ExecutionResult.md): Discriminated union for plugin execution results.
204
248
  - [Type Alias: FileAction](./docs/api/appkit/TypeAlias.FileAction.md): Every action the files plugin can perform.
205
249
  - [Type Alias: FilePolicy()](./docs/api/appkit/TypeAlias.FilePolicy.md): A policy function that decides whether user may perform action on
@@ -207,6 +251,7 @@ npx @databricks/appkit docs <query>
207
251
  - [Type Alias: IAppRouter](./docs/api/appkit/TypeAlias.IAppRouter.md): Express router type for plugin route registration
208
252
  - [Type Alias: IDatabaseConfig<TSchema>](./docs/api/appkit/TypeAlias.IDatabaseConfig.md): Configuration for one schema-bound DatabasePlugin instance.
209
253
  - [Type Alias: JobsExport()](./docs/api/appkit/TypeAlias.JobsExport.md): Public API shape of the jobs plugin.
254
+ - [Type Alias: Matcher()](./docs/api/appkit/TypeAlias.Matcher.md): A deterministic matcher: inspects a string value and returns a result.
210
255
  - [Type Alias: PluginData<T, U, N>](./docs/api/appkit/TypeAlias.PluginData.md): Tuple of plugin class, config, and name. Created by toPlugin() and passed to createApp().
211
256
  - [Type Alias: Plugins](./docs/api/appkit/TypeAlias.Plugins.md): Plugin map passed to the function form of AgentDefinition.tools.
212
257
  - [Type Alias: ResolvedToolEntry](./docs/api/appkit/TypeAlias.ResolvedToolEntry.md): Internal tool-index entry after a tool record has been resolved to a dispatchable form.
@@ -214,6 +259,7 @@ npx @databricks/appkit docs <query>
214
259
  - [Type Alias: ResourcePermission](./docs/api/appkit/TypeAlias.ResourcePermission.md): Union of all possible permission levels across all resource types.
215
260
  - [Type Alias: SearchFilters](./docs/api/appkit/TypeAlias.SearchFilters.md)
216
261
  - [Type Alias: ServingFactory](./docs/api/appkit/TypeAlias.ServingFactory.md): Factory function returned by AppKit.serving.
262
+ - [Type Alias: Severity](./docs/api/appkit/TypeAlias.Severity.md): Whether an assertion fails the eval (gate) or is tracked only (soft).
217
263
  - [Type Alias: SupervisorTool](./docs/api/appkit/TypeAlias.SupervisorTool.md): Tools supported by the Databricks AI Gateway Responses API. The shapes match
218
264
  - [Type Alias: ToolRegistry](./docs/api/appkit/TypeAlias.ToolRegistry.md)
219
265
  - [Type Alias: ToPlugin()<T, U, N>](./docs/api/appkit/TypeAlias.ToPlugin.md): Factory function type returned by toPlugin(). Accepts optional config and returns a PluginData tuple.
package/NOTICE.md CHANGED
@@ -8,6 +8,7 @@ This Software contains code from the following open source projects:
8
8
  | :--------------- | :---------------- | :----------- | :--------------------------------------------------- |
9
9
  | [@ast-grep/napi](https://www.npmjs.com/package/@ast-grep/napi) | 0.37.0 | MIT | https://ast-grep.github.io |
10
10
  | [@clack/prompts](https://www.npmjs.com/package/@clack/prompts) | 1.0.1 | MIT | https://github.com/bombshell-dev/clack/tree/main/packages/prompts#readme |
11
+ | [@mlflow/core](https://www.npmjs.com/package/@mlflow/core) | 0.4.0 | Apache-2.0 | https://mlflow.org/ |
11
12
  | [@opentelemetry/api](https://www.npmjs.com/package/@opentelemetry/api) | 1.9.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/api |
12
13
  | [@opentelemetry/api-logs](https://www.npmjs.com/package/@opentelemetry/api-logs) | 0.205.0, 0.219.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/experimental/packages/api-logs |
13
14
  | [@opentelemetry/auto-instrumentations-node](https://www.npmjs.com/package/@opentelemetry/auto-instrumentations-node) | 0.77.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js-contrib/tree/main/packages/auto-instrumentations-node#readme |
@@ -20,8 +21,8 @@ This Software contains code from the following open source projects:
20
21
  | [@opentelemetry/resources](https://www.npmjs.com/package/@opentelemetry/resources) | 2.1.0, 2.8.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/packages/opentelemetry-resources |
21
22
  | [@opentelemetry/sdk-logs](https://www.npmjs.com/package/@opentelemetry/sdk-logs) | 0.205.0, 0.219.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/experimental/packages/sdk-logs |
22
23
  | [@opentelemetry/sdk-metrics](https://www.npmjs.com/package/@opentelemetry/sdk-metrics) | 2.1.0, 2.8.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/packages/sdk-metrics |
23
- | [@opentelemetry/sdk-node](https://www.npmjs.com/package/@opentelemetry/sdk-node) | 0.205.0, 0.219.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/experimental/packages/opentelemetry-sdk-node |
24
24
  | [@opentelemetry/sdk-trace-base](https://www.npmjs.com/package/@opentelemetry/sdk-trace-base) | 2.1.0, 2.8.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/packages/opentelemetry-sdk-trace-base |
25
+ | [@opentelemetry/sdk-trace-node](https://www.npmjs.com/package/@opentelemetry/sdk-trace-node) | 2.1.0, 2.8.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/packages/opentelemetry-sdk-trace-node |
25
26
  | [@opentelemetry/semantic-conventions](https://www.npmjs.com/package/@opentelemetry/semantic-conventions) | 1.38.0 | Apache-2.0 | https://github.com/open-telemetry/opentelemetry-js/tree/main/semantic-conventions |
26
27
  | [@radix-ui/react-accordion](https://www.npmjs.com/package/@radix-ui/react-accordion) | 1.2.12 | MIT | https://radix-ui.com/primitives |
27
28
  | [@radix-ui/react-alert-dialog](https://www.npmjs.com/package/@radix-ui/react-alert-dialog) | 1.1.15 | MIT | https://radix-ui.com/primitives |
@@ -71,7 +72,6 @@ This Software contains code from the following open source projects:
71
72
  | [lucide-react](https://www.npmjs.com/package/lucide-react) | 0.554.0 | ISC | https://lucide.dev |
72
73
  | [magic-string](https://www.npmjs.com/package/magic-string) | 0.30.21 | MIT | https://github.com/Rich-Harris/magic-string#readme |
73
74
  | [marked](https://www.npmjs.com/package/marked) | 16.4.2, 17.0.3 | MIT | https://marked.js.org |
74
- | [mlflow-tracing](https://www.npmjs.com/package/mlflow-tracing) | 0.1.3 | Apache-2.0 | https://mlflow.org/ |
75
75
  | [next-themes](https://www.npmjs.com/package/next-themes) | 0.4.6 | MIT | https://github.com/pacocoursey/next-themes#readme |
76
76
  | [obug](https://www.npmjs.com/package/obug) | 2.1.1 | MIT | https://github.com/sxzz/obug#readme |
77
77
  | [pg](https://www.npmjs.com/package/pg) | 8.18.0 | MIT | https://github.com/brianc/node-postgres |
@@ -1,6 +1,6 @@
1
1
  //#region package.json
2
2
  var name = "@databricks/appkit";
3
- var version = "0.70.0";
3
+ var version = "0.72.0";
4
4
 
5
5
  //#endregion
6
6
  export { name, version };
package/dist/beta.d.ts CHANGED
@@ -15,6 +15,18 @@ import { Schema } from "./database/schema-builder/types.js";
15
15
  import { bigid, bigint, boolean, enumColumn, id, integer, jsonb, text, timestamp, uuid, varchar } from "./database/schema-builder/columns.js";
16
16
  import { defineSchema } from "./database/schema-builder/define-schema.js";
17
17
  import { fk } from "./database/schema-builder/fk.js";
18
+ import { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth } from "./connectors/mlflow/auth.js";
19
+ import { MlflowClient, PostResult, normalizeHost } from "./connectors/mlflow/client.js";
20
+ import { AssertionHandle, AssertionResult, CustomJudgeSpec, DriveResult, EvalDefinition, EvalDriver, EvalResult, MatchResult, Matcher, Severity, TestContext } from "./evals/types.js";
21
+ import { defineEval } from "./evals/define-eval.js";
22
+ import { DiscoveredEval, discoverEvalFiles } from "./evals/discover.js";
23
+ import { HttpDriverOptions, createHttpDriver } from "./evals/http-driver.js";
24
+ import { JudgeConfig, JudgeScore, configureJudge, isJudgeConfigured } from "./evals/judge.js";
25
+ import { equals, includes, matches } from "./evals/matchers.js";
26
+ import { Assessment, ReportOutcome, buildAssessments, reportToMlflow } from "./evals/mlflow-report.js";
27
+ import { EvalSummary, evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize } from "./evals/report.js";
28
+ import { RunEvalOptions, runEval } from "./evals/run-eval.js";
29
+ import { EvalProgress, EvalRunSummary, RunEvalsOptions, runEvalsInDir } from "./evals/run-evals.js";
18
30
  import { agentIdFromMarkdownPath, loadAgentFromFile, loadAgentsFromDir } from "./core/agent/load-agents.js";
19
31
  import { agents } from "./plugins/agents/agents.js";
20
32
  import "./plugins/agents/index.js";
@@ -25,4 +37,4 @@ import { IDatabaseConfig } from "./plugins/database/types.js";
25
37
  import { database } from "./plugins/database/database.js";
26
38
  import "./plugins/database/index.js";
27
39
  import "./plugins/beta-exports.generated.js";
28
- export { type AgentAdapter, type AgentDefinition, type AgentEvent, type AgentInput, type AgentRunContext, type AgentTool, type AgentToolDefinition, type AgentTools, type AgentToolsFn, type AgentsPluginConfig, AppKitMcpClient, type AutoInheritToolsConfig, type BaseSystemPromptOption, type DatabaseExports, DatabricksAdapter, type FunctionTool, type GenerationParams, type HostedSupervisorTool, type HostedTool, type IAiSearchConfig, type IDatabaseConfig, type IndexConfig, type McpConnectAllResult, type Message, type PluginToolkitProvider, type Plugins, type PromptContext, type RegisteredAgent, type RerankerConfig, type ResolvedToolEntry, type RunAgentInput, type RunAgentResult, SUPERVISOR_EXTENSION_KEY, type Schema, type SearchFilters, type SearchRequest, type SearchResponse, type SearchResult, SupervisorApiAdapter, type SupervisorApiAdapterOptions, type SupervisorExtension, type SupervisorTool, type Thread, type ThreadStore, type ToolAnnotations, type ToolConfig, type ToolEntry, type ToolProvider, type ToolRegistry, type ToolkitEntry, type ToolkitOptions, type WorkspaceClientLike, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, createAgent, database, defineSchema, defineTool, enumColumn, executeFromRegistry, fk, fromSupervisorApi, functionToolToDefinition, id, integer, isFunctionTool, isHostedTool, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, mcpServer, parseTextToolCalls, resolveHostedTools, runAgent, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
40
+ export { type AgentAdapter, type AgentDefinition, type AgentEvent, type AgentInput, type AgentRunContext, type AgentTool, type AgentToolDefinition, type AgentTools, type AgentToolsFn, type AgentsPluginConfig, AppKitMcpClient, AssertionHandle, AssertionResult, Assessment, type AutoInheritToolsConfig, type BaseSystemPromptOption, CustomJudgeSpec, type DatabaseExports, DatabricksAdapter, DatabricksAuth, DiscoveredEval, DriveResult, EvalDefinition, EvalDriver, EvalProgress, EvalResult, EvalRunSummary, EvalSummary, type FunctionTool, type GenerationParams, type HostedSupervisorTool, type HostedTool, HttpDriverOptions, type IAiSearchConfig, type IDatabaseConfig, type IndexConfig, JudgeConfig, JudgeScore, MatchResult, Matcher, type McpConnectAllResult, type Message, MlflowClient, type PluginToolkitProvider, type Plugins, PostResult, type PromptContext, type RegisteredAgent, ReportOutcome, type RerankerConfig, ResolveDatabricksAuthOptions, type ResolvedToolEntry, type RunAgentInput, type RunAgentResult, RunEvalOptions, RunEvalsOptions, SUPERVISOR_EXTENSION_KEY, type Schema, type SearchFilters, type SearchRequest, type SearchResponse, type SearchResult, Severity, SupervisorApiAdapter, type SupervisorApiAdapterOptions, type SupervisorExtension, type SupervisorTool, TestContext, type Thread, type ThreadStore, type ToolAnnotations, type ToolConfig, type ToolEntry, type ToolProvider, type ToolRegistry, type ToolkitEntry, type ToolkitOptions, type WorkspaceClientLike, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineSchema, defineTool, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, runAgent, runEval, runEvalsInDir, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
package/dist/beta.js CHANGED
@@ -1,4 +1,6 @@
1
1
  import { AppKitMcpClient } from "./connectors/mcp/client.js";
2
+ import { resolveDatabricksAuth } from "./connectors/mlflow/auth.js";
3
+ import { MlflowClient, normalizeHost } from "./connectors/mlflow/client.js";
2
4
  import { tool } from "./core/agent/tools/tool.js";
3
5
  import { defineTool, executeFromRegistry, toolsFromRegistry } from "./core/agent/tools/define-tool.js";
4
6
  import { bigid, bigint, boolean, enumColumn, id, integer, jsonb, text, timestamp, uuid, varchar } from "./database/schema-builder/columns.js";
@@ -13,6 +15,16 @@ import { isToolkitEntry } from "./core/agent/types.js";
13
15
  import { runAgent } from "./core/agent/run-agent.js";
14
16
  import "./core/agent/tools/index.js";
15
17
  import "./database/schema-builder/index.js";
18
+ import { defineEval } from "./evals/define-eval.js";
19
+ import { discoverEvalFiles } from "./evals/discover.js";
20
+ import { createHttpDriver } from "./evals/http-driver.js";
21
+ import { configureJudge, isJudgeConfigured } from "./evals/judge.js";
22
+ import { equals, includes, matches } from "./evals/matchers.js";
23
+ import { buildAssessments, reportToMlflow } from "./evals/mlflow-report.js";
24
+ import { evalGlyph, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, summarize } from "./evals/report.js";
25
+ import { runEval } from "./evals/run-eval.js";
26
+ import { runEvalsInDir } from "./evals/run-evals.js";
27
+ import "./evals/index.js";
16
28
  import { agentIdFromMarkdownPath, loadAgentFromFile, loadAgentsFromDir } from "./core/agent/load-agents.js";
17
29
  import { agents } from "./plugins/agents/agents.js";
18
30
  import "./plugins/agents/index.js";
@@ -20,4 +32,4 @@ import { aiSearch } from "./plugins/ai-search/ai-search.js";
20
32
  import { database } from "./plugins/database/database.js";
21
33
  import "./plugins/beta-exports.generated.js";
22
34
 
23
- export { AppKitMcpClient, DatabricksAdapter, SUPERVISOR_EXTENSION_KEY, SupervisorApiAdapter, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, createAgent, database, defineSchema, defineTool, enumColumn, executeFromRegistry, fk, fromSupervisorApi, functionToolToDefinition, id, integer, isFunctionTool, isHostedTool, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, mcpServer, parseTextToolCalls, resolveHostedTools, runAgent, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
35
+ export { AppKitMcpClient, DatabricksAdapter, MlflowClient, SUPERVISOR_EXTENSION_KEY, SupervisorApiAdapter, agentIdFromMarkdownPath, agents, aiSearch, bigid, bigint, boolean, buildAssessments, configureJudge, createAgent, createHttpDriver, database, defineEval, defineSchema, defineTool, discoverEvalFiles, enumColumn, equals, evalGlyph, executeFromRegistry, fk, formatEvalDetail, formatEvalHeadline, formatEvalResults, formatSummaryLine, fromSupervisorApi, functionToolToDefinition, id, includes, integer, isFunctionTool, isHostedTool, isJudgeConfigured, isSupervisorTool, isToolkitEntry, jsonb, loadAgentFromFile, loadAgentsFromDir, matches, mcpServer, normalizeHost, parseTextToolCalls, reportToMlflow, resolveDatabricksAuth, resolveHostedTools, runAgent, runEval, runEvalsInDir, summarize, supervisorTools, text, timestamp, tool, toolsFromRegistry, uuid, varchar };
@@ -0,0 +1,112 @@
1
+ import { Command } from "commander";
2
+
3
+ //#region src/cli/commands/agent/eval.ts
4
+ /**
5
+ * Loaded at runtime from the consuming project so this command (which ships in
6
+ * `@databricks/shared`) doesn't take a build-time dependency on appkit. The
7
+ * specifier is a variable so the type checker treats it as `any`.
8
+ */
9
+ async function loadRunner() {
10
+ const spec = "@databricks/appkit/beta";
11
+ try {
12
+ return await import(spec);
13
+ } catch (err) {
14
+ throw new Error(`Could not load @databricks/appkit. Run \`appkit agent eval\` from a project with @databricks/appkit installed. Cause: ${err instanceof Error ? err.message : String(err)}`);
15
+ }
16
+ }
17
+ function parseHeaders(values) {
18
+ const headers = {};
19
+ for (const v of values) {
20
+ const i = v.indexOf(":");
21
+ if (i === -1) continue;
22
+ headers[v.slice(0, i).trim()] = v.slice(i + 1).trim();
23
+ }
24
+ return headers;
25
+ }
26
+ /**
27
+ * Native MLflow "Evaluation run" config — only when creds + an experiment are
28
+ * all present (traces live in the app; the run + scores are driven from here).
29
+ */
30
+ function resolveMlflow(opts, auth) {
31
+ const experimentId = opts.experiment ?? process.env.MLFLOW_EXPERIMENT_ID;
32
+ if (!(auth.host && auth.token && experimentId)) return void 0;
33
+ const sqlWarehouseId = opts.warehouseId ?? process.env.MLFLOW_TRACING_SQL_WAREHOUSE_ID ?? process.env.DATABRICKS_WAREHOUSE_ID;
34
+ return {
35
+ host: auth.host,
36
+ token: auth.token,
37
+ experimentId,
38
+ ...sqlWarehouseId ? { sqlWarehouseId } : {}
39
+ };
40
+ }
41
+ /** LLM-as-judge config — reuses the Databricks creds + a judge serving endpoint. */
42
+ function resolveJudge(opts, auth) {
43
+ const model = opts.judgeModel ?? process.env.APPKIT_JUDGE_MODEL;
44
+ return model && auth.host && auth.token ? {
45
+ host: auth.host,
46
+ token: auth.token,
47
+ model
48
+ } : void 0;
49
+ }
50
+ /** Progress reporter: stream each eval as it runs instead of going silent. */
51
+ function makeProgressReporter(runner, url) {
52
+ return (event) => {
53
+ switch (event.type) {
54
+ case "discovered":
55
+ console.log(`Running ${event.total} eval${event.total === 1 ? "" : "s"} against ${url}\n`);
56
+ break;
57
+ case "run-created":
58
+ console.log(`MLflow evaluation run: ${event.runId}\n`);
59
+ break;
60
+ case "result":
61
+ console.log(`[${event.index + 1}/${event.total}] ${runner.formatEvalHeadline(event.result)}`);
62
+ for (const line of runner.formatEvalDetail(event.result)) console.log(line);
63
+ break;
64
+ }
65
+ };
66
+ }
67
+ function formatFailureLine(f) {
68
+ return ` ✗ trace ${f.traceId}: ${f.status ?? ""} ${f.error ?? ""}`.trim();
69
+ }
70
+ /** Print the MLflow assessment/finish outcome after a run that created one. */
71
+ function printMlflowOutcome(mlflow) {
72
+ const { report, finish } = mlflow;
73
+ console.log(`MLflow: ${report.written} assessment(s) written` + (report.skipped ? `, ${report.skipped} skipped` : "") + (report.failures.length ? `, ${report.failures.length} failed` : ""));
74
+ for (const f of report.failures) console.error(formatFailureLine(f));
75
+ if (finish.metricsError) console.error(` ⚠ metrics not logged: ${finish.metricsError}`);
76
+ if (!finish.finished) console.error(` ✗ run left RUNNING — failed to finish: ${finish.finishError ?? "unknown"}`);
77
+ }
78
+ async function runAgentEval(filter, opts) {
79
+ const runner = await loadRunner();
80
+ const auth = await runner.resolveDatabricksAuth({
81
+ profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,
82
+ host: opts.databricksHost ?? process.env.DATABRICKS_HOST,
83
+ token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN
84
+ }) ?? {};
85
+ let summary;
86
+ try {
87
+ summary = await runner.runEvalsInDir({
88
+ rootDir: opts.root,
89
+ baseUrl: opts.url,
90
+ filter,
91
+ strict: opts.strict,
92
+ headers: opts.header ? parseHeaders(opts.header) : void 0,
93
+ concurrency: opts.concurrency,
94
+ mlflow: resolveMlflow(opts, auth),
95
+ judge: resolveJudge(opts, auth),
96
+ onEvent: makeProgressReporter(runner, opts.url)
97
+ });
98
+ } catch (err) {
99
+ console.error(`\nEval run failed: ${err instanceof Error ? err.message : String(err)}`);
100
+ process.exitCode = 1;
101
+ return;
102
+ }
103
+ console.log(`\n${runner.formatSummaryLine(summary.results)}`);
104
+ if (summary.mlflow) printMlflowOutcome(summary.mlflow);
105
+ else console.log("\nMLflow evaluation run skipped — pass --experiment (or set MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.");
106
+ if (!runner.summarize(summary.results).allPassed) process.exitCode = 1;
107
+ }
108
+ const agentEvalCommand = new Command("eval").description("Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app").argument("[filter]", "Only run evals whose <agent>/<id> contains this substring (or an exact agent id)").option("--url <url>", "Base URL of the running app", "http://localhost:3000").option("--strict", "Fail on soft-assertion misses too", false).option("--concurrency <n>", "Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)", (v) => Number.parseInt(v, 10)).option("--root <dir>", "Project root containing server/agents/ (default: cwd)").option("--header <header...>", "Extra request header as 'Key: value' (repeatable)").option("--profile <name>", "Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)").option("--databricks-host <host>", "Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)").option("--databricks-token <token>", "Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)").option("--experiment <id>", "MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)").option("--warehouse-id <id>", "SQL warehouse id for writing assessments to UC-backed experiments (default: MLFLOW_TRACING_SQL_WAREHOUSE_ID or DATABRICKS_WAREHOUSE_ID)").option("--judge-model <endpoint>", "Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)").action(runAgentEval);
109
+
110
+ //#endregion
111
+ export { agentEvalCommand };
112
+ //# sourceMappingURL=eval.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"eval.js","names":[],"sources":["../../../../src/cli/commands/agent/eval.ts"],"sourcesContent":["import { Command } from \"commander\";\n\ninterface EvalRunSummary {\n results: unknown[];\n mlflow?: {\n runId: string;\n report: {\n written: number;\n skipped: number;\n failures: Array<{ traceId: string; status?: number; error?: string }>;\n };\n finish: { finished: boolean; metricsError?: string; finishError?: string };\n };\n}\n\ntype EvalProgress =\n | { type: \"discovered\"; total: number }\n | { type: \"run-created\"; runId: string }\n | { type: \"start\"; id: string; index: number; total: number }\n | { type: \"result\"; result: unknown; index: number; total: number };\n\n/** Subset of `@databricks/appkit/beta`'s eval runner used by this command. */\ninterface EvalRunner {\n runEvalsInDir(opts: {\n rootDir?: string;\n baseUrl: string;\n filter?: string;\n strict?: boolean;\n headers?: Record<string, string>;\n concurrency?: number;\n mlflow?: {\n host: string;\n token: string;\n experimentId: string;\n sqlWarehouseId?: string;\n };\n judge?: { host: string; token: string; model: string };\n onEvent?: (event: EvalProgress) => void;\n }): Promise<EvalRunSummary>;\n resolveDatabricksAuth(opts: {\n profile?: string;\n host?: string;\n token?: string;\n }): Promise<{ host: string; token: string } | undefined>;\n formatEvalHeadline(result: unknown): string;\n formatEvalDetail(result: unknown): string[];\n formatSummaryLine(results: unknown[]): string;\n summarize(results: unknown[]): { allPassed: boolean };\n}\n\n/**\n * Loaded at runtime from the consuming project so this command (which ships in\n * `@databricks/shared`) doesn't take a build-time dependency on appkit. The\n * specifier is a variable so the type checker treats it as `any`.\n */\nasync function loadRunner(): Promise<EvalRunner> {\n const spec = \"@databricks/appkit/beta\";\n try {\n return (await import(spec)) as unknown as EvalRunner;\n } catch (err) {\n throw new Error(\n \"Could not load @databricks/appkit. Run `appkit agent eval` from a \" +\n \"project with @databricks/appkit installed. \" +\n `Cause: ${err instanceof Error ? err.message : String(err)}`,\n );\n }\n}\n\nfunction parseHeaders(values: string[]): Record<string, string> {\n const headers: Record<string, string> = {};\n for (const v of values) {\n const i = v.indexOf(\":\");\n if (i === -1) continue;\n headers[v.slice(0, i).trim()] = v.slice(i + 1).trim();\n }\n return headers;\n}\n\ninterface EvalOptions {\n url: string;\n strict?: boolean;\n root?: string;\n header?: string[];\n profile?: string;\n databricksHost?: string;\n databricksToken?: string;\n experiment?: string;\n judgeModel?: string;\n concurrency?: number;\n warehouseId?: string;\n}\n\n/** Resolved Databricks host + bearer (either field may be absent). */\ntype Auth = { host?: string; token?: string };\n\n/**\n * Native MLflow \"Evaluation run\" config — only when creds + an experiment are\n * all present (traces live in the app; the run + scores are driven from here).\n */\nfunction resolveMlflow(opts: EvalOptions, auth: Auth) {\n const experimentId = opts.experiment ?? process.env.MLFLOW_EXPERIMENT_ID;\n if (!(auth.host && auth.token && experimentId)) return undefined;\n // UC-backed experiments need a SQL warehouse to write assessments to their\n // V4 traces. Mirror mlflow's env var, and accept the common DATABRICKS one.\n const sqlWarehouseId =\n opts.warehouseId ??\n process.env.MLFLOW_TRACING_SQL_WAREHOUSE_ID ??\n process.env.DATABRICKS_WAREHOUSE_ID;\n return {\n host: auth.host,\n token: auth.token,\n experimentId,\n ...(sqlWarehouseId ? { sqlWarehouseId } : {}),\n };\n}\n\n/** LLM-as-judge config — reuses the Databricks creds + a judge serving endpoint. */\nfunction resolveJudge(opts: EvalOptions, auth: Auth) {\n const model = opts.judgeModel ?? process.env.APPKIT_JUDGE_MODEL;\n return model && auth.host && auth.token\n ? { host: auth.host, token: auth.token, model }\n : undefined;\n}\n\n/** Progress reporter: stream each eval as it runs instead of going silent. */\nfunction makeProgressReporter(\n runner: EvalRunner,\n url: string,\n): (event: EvalProgress) => void {\n return (event) => {\n switch (event.type) {\n case \"discovered\":\n console.log(\n `Running ${event.total} eval${event.total === 1 ? \"\" : \"s\"} against ${url}\\n`,\n );\n break;\n case \"run-created\":\n console.log(`MLflow evaluation run: ${event.runId}\\n`);\n break;\n case \"result\": {\n // One full line per completion — evals run concurrently, so a split\n // \"start … glyph\" prefix would interleave into garbage.\n console.log(\n `[${event.index + 1}/${event.total}] ${runner.formatEvalHeadline(event.result)}`,\n );\n for (const line of runner.formatEvalDetail(event.result)) {\n console.log(line);\n }\n break;\n }\n }\n };\n}\n\nfunction formatFailureLine(f: {\n traceId: string;\n status?: number;\n error?: string;\n}): string {\n return ` ✗ trace ${f.traceId}: ${f.status ?? \"\"} ${f.error ?? \"\"}`.trim();\n}\n\n/** Print the MLflow assessment/finish outcome after a run that created one. */\nfunction printMlflowOutcome(\n mlflow: NonNullable<EvalRunSummary[\"mlflow\"]>,\n): void {\n const { report, finish } = mlflow;\n console.log(\n `MLflow: ${report.written} assessment(s) written` +\n (report.skipped ? `, ${report.skipped} skipped` : \"\") +\n (report.failures.length ? `, ${report.failures.length} failed` : \"\"),\n );\n for (const f of report.failures) {\n console.error(formatFailureLine(f));\n }\n if (finish.metricsError) {\n console.error(` ⚠ metrics not logged: ${finish.metricsError}`);\n }\n if (!finish.finished) {\n console.error(\n ` ✗ run left RUNNING — failed to finish: ${finish.finishError ?? \"unknown\"}`,\n );\n }\n}\n\nasync function runAgentEval(\n filter: string | undefined,\n opts: EvalOptions,\n): Promise<void> {\n const runner = await loadRunner();\n\n // Resolve Databricks host + bearer the AppKit-native way: an explicit\n // host/token (or DATABRICKS_* env) wins; otherwise the SDK mints an OAuth\n // token from the CLI profile — so no hand-set PAT is required.\n const auth: Auth =\n (await runner.resolveDatabricksAuth({\n profile: opts.profile ?? process.env.DATABRICKS_CONFIG_PROFILE,\n host: opts.databricksHost ?? process.env.DATABRICKS_HOST,\n token: opts.databricksToken ?? process.env.DATABRICKS_TOKEN,\n })) ?? {};\n\n let summary: EvalRunSummary;\n try {\n summary = await runner.runEvalsInDir({\n rootDir: opts.root,\n baseUrl: opts.url,\n filter,\n strict: opts.strict,\n headers: opts.header ? parseHeaders(opts.header) : undefined,\n concurrency: opts.concurrency,\n mlflow: resolveMlflow(opts, auth),\n judge: resolveJudge(opts, auth),\n onEvent: makeProgressReporter(runner, opts.url),\n });\n } catch (err) {\n // Setup failures (e.g. a bad --experiment for the MLflow run) reject before\n // any eval runs; surface a clean message + non-zero exit rather than an\n // unhandled promise rejection with a raw stack.\n console.error(\n `\\nEval run failed: ${err instanceof Error ? err.message : String(err)}`,\n );\n process.exitCode = 1;\n return;\n }\n console.log(`\\n${runner.formatSummaryLine(summary.results)}`);\n\n if (summary.mlflow) {\n printMlflowOutcome(summary.mlflow);\n } else {\n console.log(\n \"\\nMLflow evaluation run skipped — pass --experiment (or set\" +\n \" MLFLOW_EXPERIMENT_ID) plus --profile/--databricks-host to create one.\",\n );\n }\n\n if (!runner.summarize(summary.results).allPassed) {\n process.exitCode = 1;\n }\n}\n\nexport const agentEvalCommand = new Command(\"eval\")\n .description(\n \"Run agent evals (server/agents/<id>/evals/*.eval.ts) against a running app\",\n )\n .argument(\n \"[filter]\",\n \"Only run evals whose <agent>/<id> contains this substring (or an exact agent id)\",\n )\n .option(\"--url <url>\", \"Base URL of the running app\", \"http://localhost:3000\")\n .option(\"--strict\", \"Fail on soft-assertion misses too\", false)\n .option(\n \"--concurrency <n>\",\n \"Max evals to run concurrently (default 4; keep at or below the app's max concurrent streams per user)\",\n (v) => Number.parseInt(v, 10),\n )\n .option(\n \"--root <dir>\",\n \"Project root containing server/agents/ (default: cwd)\",\n )\n .option(\n \"--header <header...>\",\n \"Extra request header as 'Key: value' (repeatable)\",\n )\n .option(\n \"--profile <name>\",\n \"Databricks CLI profile to authenticate with via OAuth (default: DATABRICKS_CONFIG_PROFILE)\",\n )\n .option(\n \"--databricks-host <host>\",\n \"Databricks host for writing MLflow assessments (default: DATABRICKS_HOST)\",\n )\n .option(\n \"--databricks-token <token>\",\n \"Databricks token for writing MLflow assessments (default: DATABRICKS_TOKEN)\",\n )\n .option(\n \"--experiment <id>\",\n \"MLflow experiment id for the evaluation run (default: MLFLOW_EXPERIMENT_ID)\",\n )\n .option(\n \"--warehouse-id <id>\",\n \"SQL warehouse id for writing assessments to UC-backed experiments (default: MLFLOW_TRACING_SQL_WAREHOUSE_ID or DATABRICKS_WAREHOUSE_ID)\",\n )\n .option(\n \"--judge-model <endpoint>\",\n \"Databricks serving endpoint to use as the LLM judge for t.judge.* (default: APPKIT_JUDGE_MODEL)\",\n )\n .action(runAgentEval);\n"],"mappings":";;;;;;;;AAuDA,eAAe,aAAkC;CAC/C,MAAM,OAAO;AACb,KAAI;AACF,SAAQ,MAAM,OAAO;UACd,KAAK;AACZ,QAAM,IAAI,MACR,yHAEY,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GAC7D;;;AAIL,SAAS,aAAa,QAA0C;CAC9D,MAAM,UAAkC,EAAE;AAC1C,MAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,IAAI,EAAE,QAAQ,IAAI;AACxB,MAAI,MAAM,GAAI;AACd,UAAQ,EAAE,MAAM,GAAG,EAAE,CAAC,MAAM,IAAI,EAAE,MAAM,IAAI,EAAE,CAAC,MAAM;;AAEvD,QAAO;;;;;;AAwBT,SAAS,cAAc,MAAmB,MAAY;CACpD,MAAM,eAAe,KAAK,cAAc,QAAQ,IAAI;AACpD,KAAI,EAAE,KAAK,QAAQ,KAAK,SAAS,cAAe,QAAO;CAGvD,MAAM,iBACJ,KAAK,eACL,QAAQ,IAAI,mCACZ,QAAQ,IAAI;AACd,QAAO;EACL,MAAM,KAAK;EACX,OAAO,KAAK;EACZ;EACA,GAAI,iBAAiB,EAAE,gBAAgB,GAAG,EAAE;EAC7C;;;AAIH,SAAS,aAAa,MAAmB,MAAY;CACnD,MAAM,QAAQ,KAAK,cAAc,QAAQ,IAAI;AAC7C,QAAO,SAAS,KAAK,QAAQ,KAAK,QAC9B;EAAE,MAAM,KAAK;EAAM,OAAO,KAAK;EAAO;EAAO,GAC7C;;;AAIN,SAAS,qBACP,QACA,KAC+B;AAC/B,SAAQ,UAAU;AAChB,UAAQ,MAAM,MAAd;GACE,KAAK;AACH,YAAQ,IACN,WAAW,MAAM,MAAM,OAAO,MAAM,UAAU,IAAI,KAAK,IAAI,WAAW,IAAI,IAC3E;AACD;GACF,KAAK;AACH,YAAQ,IAAI,0BAA0B,MAAM,MAAM,IAAI;AACtD;GACF,KAAK;AAGH,YAAQ,IACN,IAAI,MAAM,QAAQ,EAAE,GAAG,MAAM,MAAM,IAAI,OAAO,mBAAmB,MAAM,OAAO,GAC/E;AACD,SAAK,MAAM,QAAQ,OAAO,iBAAiB,MAAM,OAAO,CACtD,SAAQ,IAAI,KAAK;AAEnB;;;;AAMR,SAAS,kBAAkB,GAIhB;AACT,QAAO,aAAa,EAAE,QAAQ,IAAI,EAAE,UAAU,GAAG,GAAG,EAAE,SAAS,KAAK,MAAM;;;AAI5E,SAAS,mBACP,QACM;CACN,MAAM,EAAE,QAAQ,WAAW;AAC3B,SAAQ,IACN,WAAW,OAAO,QAAQ,2BACvB,OAAO,UAAU,KAAK,OAAO,QAAQ,YAAY,OACjD,OAAO,SAAS,SAAS,KAAK,OAAO,SAAS,OAAO,WAAW,IACpE;AACD,MAAK,MAAM,KAAK,OAAO,SACrB,SAAQ,MAAM,kBAAkB,EAAE,CAAC;AAErC,KAAI,OAAO,aACT,SAAQ,MAAM,2BAA2B,OAAO,eAAe;AAEjE,KAAI,CAAC,OAAO,SACV,SAAQ,MACN,4CAA4C,OAAO,eAAe,YACnE;;AAIL,eAAe,aACb,QACA,MACe;CACf,MAAM,SAAS,MAAM,YAAY;CAKjC,MAAM,OACH,MAAM,OAAO,sBAAsB;EAClC,SAAS,KAAK,WAAW,QAAQ,IAAI;EACrC,MAAM,KAAK,kBAAkB,QAAQ,IAAI;EACzC,OAAO,KAAK,mBAAmB,QAAQ,IAAI;EAC5C,CAAC,IAAK,EAAE;CAEX,IAAI;AACJ,KAAI;AACF,YAAU,MAAM,OAAO,cAAc;GACnC,SAAS,KAAK;GACd,SAAS,KAAK;GACd;GACA,QAAQ,KAAK;GACb,SAAS,KAAK,SAAS,aAAa,KAAK,OAAO,GAAG;GACnD,aAAa,KAAK;GAClB,QAAQ,cAAc,MAAM,KAAK;GACjC,OAAO,aAAa,MAAM,KAAK;GAC/B,SAAS,qBAAqB,QAAQ,KAAK,IAAI;GAChD,CAAC;UACK,KAAK;AAIZ,UAAQ,MACN,sBAAsB,eAAe,QAAQ,IAAI,UAAU,OAAO,IAAI,GACvE;AACD,UAAQ,WAAW;AACnB;;AAEF,SAAQ,IAAI,KAAK,OAAO,kBAAkB,QAAQ,QAAQ,GAAG;AAE7D,KAAI,QAAQ,OACV,oBAAmB,QAAQ,OAAO;KAElC,SAAQ,IACN,oIAED;AAGH,KAAI,CAAC,OAAO,UAAU,QAAQ,QAAQ,CAAC,UACrC,SAAQ,WAAW;;AAIvB,MAAa,mBAAmB,IAAI,QAAQ,OAAO,CAChD,YACC,6EACD,CACA,SACC,YACA,mFACD,CACA,OAAO,eAAe,+BAA+B,wBAAwB,CAC7E,OAAO,YAAY,qCAAqC,MAAM,CAC9D,OACC,qBACA,0GACC,MAAM,OAAO,SAAS,GAAG,GAAG,CAC9B,CACA,OACC,gBACA,wDACD,CACA,OACC,wBACA,oDACD,CACA,OACC,oBACA,6FACD,CACA,OACC,4BACA,4EACD,CACA,OACC,8BACA,8EACD,CACA,OACC,qBACA,8EACD,CACA,OACC,uBACA,0IACD,CACA,OACC,4BACA,kGACD,CACA,OAAO,aAAa"}
@@ -0,0 +1,18 @@
1
+ import { agentEvalCommand } from "./eval.js";
2
+ import { Command } from "commander";
3
+
4
+ //#region src/cli/commands/agent/index.ts
5
+ /**
6
+ * Parent command for agent development operations.
7
+ * Subcommands:
8
+ * - eval: Run agent evals against a running app
9
+ */
10
+ const agentCommand = new Command("agent").description("Agent development commands").addCommand(agentEvalCommand).addHelpText("after", `
11
+ Examples:
12
+ $ appkit agent eval
13
+ $ appkit agent eval support --strict
14
+ $ appkit agent eval --url https://my-app.databricksapps.com`);
15
+
16
+ //#endregion
17
+ export { agentCommand };
18
+ //# sourceMappingURL=index.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.js","names":[],"sources":["../../../../src/cli/commands/agent/index.ts"],"sourcesContent":["import { Command } from \"commander\";\n\nimport { agentEvalCommand } from \"./eval\";\n\n/**\n * Parent command for agent development operations.\n * Subcommands:\n * - eval: Run agent evals against a running app\n */\nexport const agentCommand = new Command(\"agent\")\n .description(\"Agent development commands\")\n .addCommand(agentEvalCommand)\n .addHelpText(\n \"after\",\n `\nExamples:\n $ appkit agent eval\n $ appkit agent eval support --strict\n $ appkit agent eval --url https://my-app.databricksapps.com`,\n );\n"],"mappings":";;;;;;;;;AASA,MAAa,eAAe,IAAI,QAAQ,QAAQ,CAC7C,YAAY,6BAA6B,CACzC,WAAW,iBAAiB,CAC5B,YACC,SACA;;;;+DAKD"}
package/dist/cli/index.js CHANGED
@@ -1,4 +1,5 @@
1
1
  #!/usr/bin/env node
2
+ import { agentCommand } from "./commands/agent/index.js";
2
3
  import { codemodCommand } from "./commands/codemod/index.js";
3
4
  import { docsCommand } from "./commands/docs.js";
4
5
  import { doctorCommand } from "./commands/doctor/index.js";
@@ -28,6 +29,7 @@ cmd.addCommand(codemodCommand);
28
29
  cmd.addCommand(doctorCommand);
29
30
  cmd.addCommand(registryCommand, { hidden: true });
30
31
  cmd.addCommand(addCommand, { hidden: true });
32
+ cmd.addCommand(agentCommand);
31
33
  await cmd.parseAsync();
32
34
 
33
35
  //#endregion
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","names":[],"sources":["../../src/cli/index.ts"],"sourcesContent":["#!/usr/bin/env node\nimport \"dotenv/config\";\nimport { readFileSync } from \"node:fs\";\nimport { dirname, join } from \"node:path\";\nimport { fileURLToPath } from \"node:url\";\n\nimport { Command } from \"commander\";\n\nimport { codemodCommand } from \"./commands/codemod/index.js\";\nimport { docsCommand } from \"./commands/docs.js\";\nimport { doctorCommand } from \"./commands/doctor/index.js\";\nimport { generateTypesCommand } from \"./commands/generate-types.js\";\nimport { lintCommand } from \"./commands/lint.js\";\nimport { pluginCommand } from \"./commands/plugin/index.js\";\nimport { addCommand } from \"./commands/registry/add.js\";\nimport { registryCommand } from \"./commands/registry/index.js\";\nimport { setupCommand } from \"./commands/setup.js\";\n\nconst __dirname = dirname(fileURLToPath(import.meta.url));\nconst pkgPath = join(__dirname, \"../../package.json\");\nconst pkg = JSON.parse(readFileSync(pkgPath, \"utf-8\"));\n\nconst cmd = new Command();\n\ncmd\n .name(\"appkit\")\n .description(\"CLI tools for Databricks AppKit\")\n .version(pkg.version);\n\ncmd.addCommand(setupCommand);\ncmd.addCommand(generateTypesCommand);\ncmd.addCommand(lintCommand);\ncmd.addCommand(docsCommand);\ncmd.addCommand(pluginCommand);\ncmd.addCommand(codemodCommand);\ncmd.addCommand(doctorCommand);\n// Registry commands are executable but hidden from --help while the feature\n// is still in development (registry + add work end-to-end but aren't announced).\ncmd.addCommand(registryCommand, { hidden: true });\ncmd.addCommand(addCommand, { hidden: true });\n\nawait cmd.parseAsync();\n"],"mappings":";;;;;;;;;;;;;;;;;AAmBA,MAAM,UAAU,KADE,QAAQ,cAAc,OAAO,KAAK,IAAI,CAAC,EACzB,qBAAqB;AACrD,MAAM,MAAM,KAAK,MAAM,aAAa,SAAS,QAAQ,CAAC;AAEtD,MAAM,MAAM,IAAI,SAAS;AAEzB,IACG,KAAK,SAAS,CACd,YAAY,kCAAkC,CAC9C,QAAQ,IAAI,QAAQ;AAEvB,IAAI,WAAW,aAAa;AAC5B,IAAI,WAAW,qBAAqB;AACpC,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,cAAc;AAC7B,IAAI,WAAW,eAAe;AAC9B,IAAI,WAAW,cAAc;AAG7B,IAAI,WAAW,iBAAiB,EAAE,QAAQ,MAAM,CAAC;AACjD,IAAI,WAAW,YAAY,EAAE,QAAQ,MAAM,CAAC;AAE5C,MAAM,IAAI,YAAY"}
1
+ {"version":3,"file":"index.js","names":[],"sources":["../../src/cli/index.ts"],"sourcesContent":["#!/usr/bin/env node\nimport \"dotenv/config\";\nimport { readFileSync } from \"node:fs\";\nimport { dirname, join } from \"node:path\";\nimport { fileURLToPath } from \"node:url\";\n\nimport { Command } from \"commander\";\n\nimport { agentCommand } from \"./commands/agent/index.js\";\nimport { codemodCommand } from \"./commands/codemod/index.js\";\nimport { docsCommand } from \"./commands/docs.js\";\nimport { doctorCommand } from \"./commands/doctor/index.js\";\nimport { generateTypesCommand } from \"./commands/generate-types.js\";\nimport { lintCommand } from \"./commands/lint.js\";\nimport { pluginCommand } from \"./commands/plugin/index.js\";\nimport { addCommand } from \"./commands/registry/add.js\";\nimport { registryCommand } from \"./commands/registry/index.js\";\nimport { setupCommand } from \"./commands/setup.js\";\n\nconst __dirname = dirname(fileURLToPath(import.meta.url));\nconst pkgPath = join(__dirname, \"../../package.json\");\nconst pkg = JSON.parse(readFileSync(pkgPath, \"utf-8\"));\n\nconst cmd = new Command();\n\ncmd\n .name(\"appkit\")\n .description(\"CLI tools for Databricks AppKit\")\n .version(pkg.version);\n\ncmd.addCommand(setupCommand);\ncmd.addCommand(generateTypesCommand);\ncmd.addCommand(lintCommand);\ncmd.addCommand(docsCommand);\ncmd.addCommand(pluginCommand);\ncmd.addCommand(codemodCommand);\ncmd.addCommand(doctorCommand);\n// Registry commands are executable but hidden from --help while the feature\n// is still in development (registry + add work end-to-end but aren't announced).\ncmd.addCommand(registryCommand, { hidden: true });\ncmd.addCommand(addCommand, { hidden: true });\ncmd.addCommand(agentCommand);\n\nawait cmd.parseAsync();\n"],"mappings":";;;;;;;;;;;;;;;;;;AAoBA,MAAM,UAAU,KADE,QAAQ,cAAc,OAAO,KAAK,IAAI,CAAC,EACzB,qBAAqB;AACrD,MAAM,MAAM,KAAK,MAAM,aAAa,SAAS,QAAQ,CAAC;AAEtD,MAAM,MAAM,IAAI,SAAS;AAEzB,IACG,KAAK,SAAS,CACd,YAAY,kCAAkC,CAC9C,QAAQ,IAAI,QAAQ;AAEvB,IAAI,WAAW,aAAa;AAC5B,IAAI,WAAW,qBAAqB;AACpC,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,YAAY;AAC3B,IAAI,WAAW,cAAc;AAC7B,IAAI,WAAW,eAAe;AAC9B,IAAI,WAAW,cAAc;AAG7B,IAAI,WAAW,iBAAiB,EAAE,QAAQ,MAAM,CAAC;AACjD,IAAI,WAAW,YAAY,EAAE,QAAQ,MAAM,CAAC;AAC5C,IAAI,WAAW,aAAa;AAE5B,MAAM,IAAI,YAAY"}
@@ -13,6 +13,8 @@ import "./jobs/index.js";
13
13
  import { buildMcpHostPolicy } from "./mcp/host-policy.js";
14
14
  import { AppKitMcpClient } from "./mcp/client.js";
15
15
  import "./mcp/index.js";
16
+ import { resolveDatabricksAuth } from "./mlflow/auth.js";
17
+ import { MlflowClient, normalizeHost } from "./mlflow/client.js";
16
18
  import { DEFAULT_WAREHOUSE_STARTUP_TIMEOUT_MS, SQLWarehouseConnector } from "./sql-warehouse/client.js";
17
19
  import "./sql-warehouse/index.js";
18
20
 
@@ -0,0 +1,18 @@
1
+ //#region src/connectors/mlflow/auth.d.ts
2
+ /** Resolved Databricks host + bearer token for the eval runner's REST calls. */
3
+ interface DatabricksAuth {
4
+ host: string;
5
+ token: string;
6
+ }
7
+ interface ResolveDatabricksAuthOptions {
8
+ /** `~/.databrickscfg` profile to authenticate with (e.g. `dogfood`). */
9
+ profile?: string;
10
+ /** Explicit host; wins over the profile/SDK-resolved host when set. */
11
+ host?: string;
12
+ /** Explicit bearer token; when set, no OAuth is minted (PAT/CI path). */
13
+ token?: string;
14
+ }
15
+ declare function resolveDatabricksAuth(options?: ResolveDatabricksAuthOptions): Promise<DatabricksAuth | undefined>;
16
+ //#endregion
17
+ export { DatabricksAuth, ResolveDatabricksAuthOptions, resolveDatabricksAuth };
18
+ //# sourceMappingURL=auth.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"auth.d.ts","names":[],"sources":["../../../src/connectors/mlflow/auth.ts"],"mappings":";;UAGiB,cAAA;EACf,IAAA;EACA,KAAA;AAAA;AAAA,UAGe,4BAAA;EAAA;EAEf,OAAA;;EAEA,IAAA;EAFA;EAIA,KAAA;AAAA;AAAA,iBA+CoB,qBAAA,CACpB,OAAA,GAAS,4BAAA,GACR,OAAA,CAAQ,cAAA"}
@@ -0,0 +1,50 @@
1
+ import { createWorkspaceClient } from "../../shared/src/workspace-client/factory.js";
2
+
3
+ //#region src/connectors/mlflow/auth.ts
4
+ /**
5
+ * Resolve `{host, token}` for the eval runner the same way the rest of AppKit
6
+ * authenticates: construct a Databricks `WorkspaceClient` and let its config
7
+ * mint (and later refresh) an OAuth bearer from the CLI profile — no hand-set
8
+ * PAT required. An explicit host/token still wins (PAT or CI env), so the SDK
9
+ * is only consulted for whatever isn't supplied.
10
+ *
11
+ * Returns `undefined` when neither an explicit token nor a resolvable profile
12
+ * yields a bearer, so the caller can treat auth as simply unavailable.
13
+ */
14
+ /** Pull the bearer out of an `Authorization: Bearer <token>` header. */
15
+ function extractBearer(headers) {
16
+ return headers.get("authorization")?.replace(/^Bearer\s+/i, "");
17
+ }
18
+ /**
19
+ * Resolve `{host, token}` via the SDK: construct a `WorkspaceClient`, let it
20
+ * mint/refresh an OAuth bearer from the profile (or reuse a PAT), and fall back
21
+ * to any explicit host/token the caller supplied. Returns `undefined` when
22
+ * either is missing or the SDK can't resolve credentials.
23
+ */
24
+ async function resolveViaSdk(options) {
25
+ try {
26
+ const client = createWorkspaceClient(options.profile ? { profile: options.profile } : {});
27
+ const headers = new Headers();
28
+ await client.config.authenticate(headers);
29
+ const token = options.token ?? extractBearer(headers);
30
+ const host = options.host ?? (await client.config.getHost()).toString().replace(/\/+$/, "");
31
+ if (!token || !host) return void 0;
32
+ return {
33
+ host,
34
+ token
35
+ };
36
+ } catch {
37
+ return;
38
+ }
39
+ }
40
+ async function resolveDatabricksAuth(options = {}) {
41
+ if (options.host && options.token) return {
42
+ host: options.host,
43
+ token: options.token
44
+ };
45
+ return resolveViaSdk(options);
46
+ }
47
+
48
+ //#endregion
49
+ export { resolveDatabricksAuth };
50
+ //# sourceMappingURL=auth.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"auth.js","names":[],"sources":["../../../src/connectors/mlflow/auth.ts"],"sourcesContent":["import { createWorkspaceClient } from \"../../workspace-client\";\n\n/** Resolved Databricks host + bearer token for the eval runner's REST calls. */\nexport interface DatabricksAuth {\n host: string;\n token: string;\n}\n\nexport interface ResolveDatabricksAuthOptions {\n /** `~/.databrickscfg` profile to authenticate with (e.g. `dogfood`). */\n profile?: string;\n /** Explicit host; wins over the profile/SDK-resolved host when set. */\n host?: string;\n /** Explicit bearer token; when set, no OAuth is minted (PAT/CI path). */\n token?: string;\n}\n\n/**\n * Resolve `{host, token}` for the eval runner the same way the rest of AppKit\n * authenticates: construct a Databricks `WorkspaceClient` and let its config\n * mint (and later refresh) an OAuth bearer from the CLI profile — no hand-set\n * PAT required. An explicit host/token still wins (PAT or CI env), so the SDK\n * is only consulted for whatever isn't supplied.\n *\n * Returns `undefined` when neither an explicit token nor a resolvable profile\n * yields a bearer, so the caller can treat auth as simply unavailable.\n */\n/** Pull the bearer out of an `Authorization: Bearer <token>` header. */\nfunction extractBearer(headers: Headers): string | undefined {\n return headers.get(\"authorization\")?.replace(/^Bearer\\s+/i, \"\");\n}\n\n/**\n * Resolve `{host, token}` via the SDK: construct a `WorkspaceClient`, let it\n * mint/refresh an OAuth bearer from the profile (or reuse a PAT), and fall back\n * to any explicit host/token the caller supplied. Returns `undefined` when\n * either is missing or the SDK can't resolve credentials.\n */\nasync function resolveViaSdk(\n options: ResolveDatabricksAuthOptions,\n): Promise<DatabricksAuth | undefined> {\n try {\n const client = createWorkspaceClient(\n options.profile ? { profile: options.profile } : {},\n );\n const headers = new Headers();\n // Mints the OAuth access token (or reuses a PAT from the profile) and adds\n // an `Authorization: Bearer <token>` header — the same call the connectors\n // use before each request.\n await client.config.authenticate(headers);\n const token = options.token ?? extractBearer(headers);\n const host =\n options.host ??\n (await client.config.getHost()).toString().replace(/\\/+$/, \"\");\n if (!token || !host) return undefined;\n return { host, token };\n } catch {\n return undefined;\n }\n}\n\nexport async function resolveDatabricksAuth(\n options: ResolveDatabricksAuthOptions = {},\n): Promise<DatabricksAuth | undefined> {\n // Fully explicit — no need to touch the SDK.\n if (options.host && options.token) {\n return { host: options.host, token: options.token };\n }\n return resolveViaSdk(options);\n}\n"],"mappings":";;;;;;;;;;;;;;AA4BA,SAAS,cAAc,SAAsC;AAC3D,QAAO,QAAQ,IAAI,gBAAgB,EAAE,QAAQ,eAAe,GAAG;;;;;;;;AASjE,eAAe,cACb,SACqC;AACrC,KAAI;EACF,MAAM,SAAS,sBACb,QAAQ,UAAU,EAAE,SAAS,QAAQ,SAAS,GAAG,EAAE,CACpD;EACD,MAAM,UAAU,IAAI,SAAS;AAI7B,QAAM,OAAO,OAAO,aAAa,QAAQ;EACzC,MAAM,QAAQ,QAAQ,SAAS,cAAc,QAAQ;EACrD,MAAM,OACJ,QAAQ,SACP,MAAM,OAAO,OAAO,SAAS,EAAE,UAAU,CAAC,QAAQ,QAAQ,GAAG;AAChE,MAAI,CAAC,SAAS,CAAC,KAAM,QAAO;AAC5B,SAAO;GAAE;GAAM;GAAO;SAChB;AACN;;;AAIJ,eAAsB,sBACpB,UAAwC,EAAE,EACL;AAErC,KAAI,QAAQ,QAAQ,QAAQ,MAC1B,QAAO;EAAE,MAAM,QAAQ;EAAM,OAAO,QAAQ;EAAO;AAErD,QAAO,cAAc,QAAQ"}