@dbx-tools/appkit-mastra 0.1.25 → 0.1.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,206 @@
1
+ /**
2
+ * Server-side Server-Sent-Events frame interceptor for the native
3
+ * Mastra agent `/stream` endpoint.
4
+ *
5
+ * The Mastra stream (`@mastra/express`) emits one `data: <json>\n\n`
6
+ * SSE frame per chunk, where the decoded JSON carries a
7
+ * discriminating `type` (`text-delta`, `reasoning-delta`,
8
+ * `tool-call`, `step-finish`, ...) and is terminated by a
9
+ * `data: [DONE]` sentinel. {@link installStreamEventInterceptor} wraps
10
+ * an Express response and runs a caller-supplied
11
+ * {@link StreamFrameInterceptor} over each decoded frame so the caller
12
+ * can keep, rewrite, or drop it; every non-`data:` byte (SSE
13
+ * keep-alive comments, the `[DONE]` sentinel, blank padding) passes
14
+ * through verbatim.
15
+ *
16
+ * The wrap is inert unless the response turns out to be a streamed
17
+ * `200 text/event-stream` body, so non-streaming JSON routes and
18
+ * error responses are never touched.
19
+ */
20
+
21
+ import { StringDecoder } from "node:string_decoder";
22
+
23
+ import type express from "express";
24
+
25
+ /** SSE event separator used by the Mastra stream wire format. */
26
+ const FRAME_DELIMITER = "\n\n";
27
+
28
+ /**
29
+ * Matches a whole SSE frame whose `data:` payload is a JSON object,
30
+ * capturing it in three parts: the `data:` field prefix (including
31
+ * whatever inter-token whitespace the producer used), the JSON object
32
+ * body, and any trailing whitespace. Anchored end-to-end so it only
33
+ * matches a single self-contained object frame; anything else (SSE
34
+ * comments, the `[DONE]` sentinel, text/array payloads) fails the
35
+ * match and is forwarded verbatim without a parse attempt. On rewrite
36
+ * the captured prefix/suffix are re-emitted as-is so only the body
37
+ * changes.
38
+ */
39
+ const DATA_OBJECT_RE = /^(data:\s*)(\{[\s\S]*\})(\s*)$/;
40
+
41
+ /**
42
+ * Per-frame interceptor invoked with the `JSON.parse`d object of every
43
+ * `data: {json}` frame on the Mastra `/stream` response. The return
44
+ * value decides the frame's fate:
45
+ *
46
+ * - `true`: forward the frame's original bytes unchanged (no
47
+ * re-serialization).
48
+ * - `false`: drop the frame entirely.
49
+ * - `{ replace }`: re-emit the frame with the original `data:` prefix
50
+ * and trailing whitespace preserved and only the JSON body replaced
51
+ * by `JSON.stringify(replace)`.
52
+ */
53
+ export type StreamFrameInterceptor = (
54
+ chunk: unknown,
55
+ ) => { replace: unknown } | true | false;
56
+
57
+ /**
58
+ * True when `res` is a `200 text/event-stream` body - the only
59
+ * response shape we intercept. Content-type based so it tracks the
60
+ * Mastra `/stream` route without hard-coding the path, and naturally
61
+ * skips the JSON routes (history, models, charts, statements) and any
62
+ * non-200 error response.
63
+ */
64
+ function isEventStream(res: express.Response): boolean {
65
+ if (res.statusCode !== 200) return false;
66
+ const contentType = String(res.getHeader("content-type") ?? "").toLowerCase();
67
+ return contentType.includes("text/event-stream");
68
+ }
69
+
70
+ /**
71
+ * Run `interceptor` over a single SSE frame. Forwards anything that
72
+ * isn't a parseable `data: {json}` object frame verbatim (comments,
73
+ * the `[DONE]` sentinel, blank padding, unparseable / non-object
74
+ * payloads) so the interceptor only ever sees real chunk objects.
75
+ * Returns the frame string to emit, or `undefined` to drop the frame.
76
+ */
77
+ function interceptFrame(
78
+ frame: string,
79
+ interceptor: StreamFrameInterceptor,
80
+ ): string | undefined {
81
+ const match = DATA_OBJECT_RE.exec(frame);
82
+ if (!match) return frame;
83
+ const [, prefix, body, suffix] = match;
84
+ let parsed: unknown;
85
+ try {
86
+ parsed = JSON.parse(body!);
87
+ } catch {
88
+ // Forward unparseable payloads untouched - feeding a partial /
89
+ // malformed frame to the interceptor risks corrupting bytes the
90
+ // client could still have handled.
91
+ return frame;
92
+ }
93
+ const result = interceptor(parsed);
94
+ if (result === true) return frame;
95
+ if (result === false) return undefined;
96
+ // Preserve the exact captured prefix/suffix so only the JSON body
97
+ // changes; the framing the client parses stays byte-identical.
98
+ return `${prefix}${JSON.stringify(result.replace)}${suffix}`;
99
+ }
100
+
101
+ /**
102
+ * Wrap `res.write` / `res.end` so each SSE frame on a streamed Mastra
103
+ * response is run through `interceptor` (keep / rewrite / drop).
104
+ *
105
+ * Interception only engages once the response is confirmed to be a
106
+ * `200 text/event-stream` body (decided lazily on the first write,
107
+ * after the handler has set status + content-type); any other
108
+ * response streams through unchanged. A {@link StringDecoder}
109
+ * reassembles multi-byte UTF-8 sequences that straddle chunk
110
+ * boundaries, and a running buffer holds the trailing partial frame
111
+ * until its `\n\n` arrives.
112
+ */
113
+ export function installStreamEventInterceptor(
114
+ res: express.Response,
115
+ interceptor: StreamFrameInterceptor,
116
+ ): void {
117
+ const origWrite = res.write.bind(res);
118
+ const origEnd = res.end.bind(res);
119
+ const decoder = new StringDecoder("utf8");
120
+ let engaged: boolean | undefined;
121
+ let buffer = "";
122
+
123
+ // Append decoded text, emit every whole `\n\n`-delimited frame the
124
+ // interceptor keeps, and retain any trailing partial frame in
125
+ // `buffer`. On `final`, the residual buffer is flushed as the last
126
+ // frame so a stream that doesn't end on a delimiter isn't dropped.
127
+ const intercept = (text: string, final: boolean): string => {
128
+ buffer += text;
129
+
130
+ const out: string[] = [];
131
+
132
+ let idx: number;
133
+ while ((idx = buffer.indexOf(FRAME_DELIMITER)) !== -1) {
134
+ const frame = buffer.slice(0, idx);
135
+ buffer = buffer.slice(idx + FRAME_DELIMITER.length);
136
+
137
+ const kept = interceptFrame(frame, interceptor);
138
+ if (kept !== undefined) {
139
+ out.push(kept, FRAME_DELIMITER);
140
+ }
141
+ }
142
+
143
+ if (final && buffer.length > 0) {
144
+ const kept = interceptFrame(buffer, interceptor);
145
+ if (kept !== undefined) out.push(kept);
146
+ buffer = "";
147
+ }
148
+
149
+ return out.length === 0 ? "" : out.join("");
150
+ };
151
+
152
+ const toText = (chunk: unknown): string => {
153
+ if (typeof chunk === "string") return chunk;
154
+ if (Buffer.isBuffer(chunk)) return decoder.write(chunk);
155
+ if (chunk instanceof Uint8Array) {
156
+ return decoder.write(
157
+ Buffer.from(chunk.buffer, chunk.byteOffset, chunk.byteLength),
158
+ );
159
+ }
160
+ return "";
161
+ };
162
+
163
+ res.write = function patchedWrite(
164
+ chunk: unknown,
165
+ encodingOrCb?: BufferEncoding | ((error?: Error | null) => void),
166
+ cb?: (error?: Error | null) => void,
167
+ ): boolean {
168
+ if (engaged === undefined) engaged = isEventStream(res);
169
+ if (!engaged) {
170
+ return origWrite(chunk as never, encodingOrCb as never, cb as never);
171
+ }
172
+ const callback = typeof encodingOrCb === "function" ? encodingOrCb : cb;
173
+ const out = intercept(toText(chunk), false);
174
+ if (out.length === 0) {
175
+ // The chunk only advanced a partial frame; nothing to forward
176
+ // yet. Honor the write callback so backpressure-aware writers
177
+ // don't stall waiting on an ack that never comes.
178
+ if (callback) queueMicrotask(() => callback());
179
+ return true;
180
+ }
181
+ return origWrite(out, callback as never);
182
+ } as typeof res.write;
183
+
184
+ res.end = function patchedEnd(
185
+ chunk?: unknown | (() => void),
186
+ encodingOrCb?: BufferEncoding | (() => void),
187
+ cb?: () => void,
188
+ ): express.Response {
189
+ if (engaged === undefined) engaged = isEventStream(res);
190
+ if (!engaged) {
191
+ return origEnd(chunk as never, encodingOrCb as never, cb as never);
192
+ }
193
+ const callback =
194
+ typeof chunk === "function"
195
+ ? chunk
196
+ : typeof encodingOrCb === "function"
197
+ ? encodingOrCb
198
+ : cb;
199
+ const text =
200
+ (typeof chunk !== "function" && chunk !== undefined ? toText(chunk) : "") +
201
+ decoder.end();
202
+ const out = intercept(text, true);
203
+ if (out.length > 0) origWrite(out);
204
+ return origEnd(callback as never);
205
+ } as typeof res.end;
206
+ }
package/src/plugin.ts CHANGED
@@ -37,25 +37,38 @@ import {
37
37
  type PluginManifest,
38
38
  type ResourceRequirement,
39
39
  } from "@databricks/appkit";
40
- import { appkitUtils, logUtils } from "@dbx-tools/shared";
40
+ import { appkitUtils, commonUtils, logUtils } from "@dbx-tools/shared";
41
41
  import { chatRoute } from "@mastra/ai-sdk";
42
42
  import type { Agent } from "@mastra/core/agent";
43
43
  import { Mastra } from "@mastra/core/mastra";
44
44
  import express from "express";
45
45
 
46
46
  import { buildAgents, FALLBACK_AGENT_ID, type BuiltAgents } from "./agents.js";
47
- import type { MastraClientConfig } from "@dbx-tools/appkit-mastra-shared";
47
+ import type {
48
+ MastraClientConfig,
49
+ StatementData,
50
+ } from "@dbx-tools/appkit-mastra-shared";
51
+ import { fetchChart } from "./chart.js";
48
52
  import type { MastraPluginConfig } from "./config.js";
49
53
  import { historyRoute } from "./history.js";
50
54
  import { createMemoryBuilder, needsLakebase } from "./memory.js";
51
55
  import { buildObservability } from "./observability.js";
52
56
  import { attachRoutePatchMiddleware, MastraServer } from "./server.js";
57
+ import {
58
+ installStreamEventInterceptor,
59
+ type StreamFrameInterceptor,
60
+ } from "./intercept.js";
53
61
  import {
54
62
  clearServingEndpointsCache,
55
63
  listServingEndpoints,
56
64
  resolveServingConfig,
57
65
  type ServingEndpointSummary,
58
66
  } from "./serving.js";
67
+ import {
68
+ fetchStatementData,
69
+ isStatementNotFoundError,
70
+ STATEMENT_ROW_CAP,
71
+ } from "./statement.js";
59
72
 
60
73
  const GENIE_MANIFEST = appkitUtils.data(genie).plugin.manifest;
61
74
  const LAKEBASE_MANIFEST = appkitUtils.data(lakebase).plugin.manifest;
@@ -199,6 +212,7 @@ export class MastraPlugin extends Plugin<MastraPluginConfig> {
199
212
  modelsPath: `${basePath}/models`,
200
213
  historyPath: `${basePath}/route/history`,
201
214
  historyPathTemplate: `${basePath}/route/history/:agentId`,
215
+ embedPathTemplate: `${basePath}/embed/:type/:id`,
202
216
  defaultAgent: this.built?.defaultAgentId ?? FALLBACK_AGENT_ID,
203
217
  agents: Object.keys(this.built?.agents ?? {}),
204
218
  };
@@ -218,12 +232,130 @@ export class MastraPlugin extends Plugin<MastraPluginConfig> {
218
232
  .catch(next);
219
233
  });
220
234
 
235
+ // `GET /embed/:type/:id` is the single resolver for every embed
236
+ // marker the agent emits in prose (`[chart:<id>]`,
237
+ // `[data:<id>]`, ...). `:type` selects a resolver from the
238
+ // registry below; `:id` is that resolver's lookup key. The
239
+ // grammar (see `marker.ts`) is type-agnostic on purpose - new
240
+ // embed kinds are added by registering a resolver here, with no
241
+ // client or grammar change.
242
+ //
243
+ // Status codes:
244
+ // - 200 with the resolver's JSON body when the id resolves.
245
+ // - 404 when `:type` isn't registered (unsupported embed
246
+ // type) OR a registered resolver can't find `:id` (unknown
247
+ // / expired - e.g. a chart past its 1h TTL or a fabricated
248
+ // id the model never minted).
249
+ // - 400 when `:id` is empty.
250
+ //
251
+ // Per-type query knobs and behavior:
252
+ // - `chart`: long-polls the chart cache until the entry
253
+ // settles (`result` / `error`) or the budget elapses (then
254
+ // returns the still-processing entry to poll again).
255
+ // `?timeoutMs=<n>` (default 60s, capped 5min) tunes it.
256
+ // - `data`: one OBO-scoped Statement Execution fetch.
257
+ // `?limit=<n>` caps rows (clamped to STATEMENT_ROW_CAP).
258
+ //
259
+ // Built once (this handler is registered once) and keyed by the
260
+ // raw `:type` token. Each resolver gets the request (for query
261
+ // parsing + OBO scoping) and an `AbortSignal` bridged off the
262
+ // connection `close` event so a long-poll unblocks the instant
263
+ // the client disconnects. `undefined` from a resolver maps to a
264
+ // clean 404; thrown errors bubble through `next(err)`.
265
+ const embedResolvers: Record<string, EmbedResolver> = {
266
+ chart: (req, id, signal) => {
267
+ const timeoutMs = parseTimeoutMs(req.query["timeoutMs"]);
268
+ return fetchChart(id, {
269
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
270
+ signal,
271
+ });
272
+ },
273
+ data: (req, id, signal) => {
274
+ const limit = parseStatementLimit(req.query["limit"]);
275
+ return this.userScopedSelf(req).fetchStatement(id, {
276
+ ...(limit !== undefined ? { limit } : {}),
277
+ signal,
278
+ });
279
+ },
280
+ };
281
+
282
+ router.get("/embed/:type/:id", (req, res, next) => {
283
+ const type = req.params["type"] ?? "";
284
+ const id = req.params["id"];
285
+ const resolve = embedResolvers[type];
286
+ if (!resolve) {
287
+ res.status(404).json({ error: `unsupported embed type: ${type}` });
288
+ return;
289
+ }
290
+ if (!id) {
291
+ res.status(400).json({ error: "id is required" });
292
+ return;
293
+ }
294
+ // Express's `req` predates `AbortSignal`; bridge the `close`
295
+ // event onto an `AbortController` so a closed connection
296
+ // unblocks any long-poll immediately and frees the request
297
+ // thread. The listener is GC'd with the request on normal
298
+ // completion.
299
+ const controller = new AbortController();
300
+ req.on("close", () => controller.abort());
301
+ resolve(req, id, controller.signal)
302
+ .then((entry) => {
303
+ if (entry === undefined) {
304
+ res.status(404).json({ error: `${type} not found` });
305
+ return;
306
+ }
307
+ res.json(entry);
308
+ })
309
+ .catch(next);
310
+ });
311
+
221
312
  router.use("", (req, res, next) => {
222
313
  if (!this.mastraApp) return res.status(503).end();
314
+ // Run each Mastra `/stream` SSE frame through the interceptor
315
+ // (keep / rewrite / drop). Only engages on a
316
+ // `200 text/event-stream` body, so the JSON routes above and any
317
+ // error response are untouched.
318
+ installStreamEventInterceptor(res, interceptStreamFrame);
223
319
  return this.userScopedSelf(req).mastraApp!(req, res, next);
224
320
  });
225
321
  }
226
322
 
323
+ /**
324
+ * Implementation backing the `data` embed resolver
325
+ * (`GET /embed/data/:id`). Runs inside the AppKit user-context proxy so
326
+ * `getExecutionContext()` returns the OBO-scoped workspace
327
+ * client, then reuses the same `fetchStatementData` pipeline
328
+ * the `get_statement` tool runs so the LLM and the UI see the
329
+ * exact same shape for the same statement.
330
+ *
331
+ * Returns `undefined` for upstream 404s so the route can map
332
+ * them to a clean HTTP 404; any other failure bubbles up.
333
+ */
334
+ private async fetchStatement(
335
+ statementId: string,
336
+ options: { limit?: number; signal?: AbortSignal } = {},
337
+ ): Promise<StatementData | undefined> {
338
+ const client = getExecutionContext().client;
339
+ const limit = Math.min(options.limit ?? STATEMENT_ROW_CAP, STATEMENT_ROW_CAP);
340
+ try {
341
+ const data = await fetchStatementData(client, statementId, {
342
+ limit,
343
+ ...(options.signal ? { signal: options.signal } : {}),
344
+ });
345
+ return {
346
+ columns: data.columns,
347
+ rows: data.rows,
348
+ rowCount: data.rowCount,
349
+ truncated: data.rows.length < data.rowCount,
350
+ };
351
+ } catch (err) {
352
+ // The Databricks SDK throws on 404; surface as `undefined`
353
+ // so the route maps to a clean HTTP 404 instead of a 500.
354
+ if (isStatementNotFoundError(err)) return undefined;
355
+ throw err;
356
+ }
357
+ }
358
+
227
359
  /**
228
360
  * Return `this.asUser(req)` when the request carries an OBO token,
229
361
  * otherwise return `this` directly. Prevents the noisy AppKit warn
@@ -326,4 +458,80 @@ export class MastraPlugin extends Plugin<MastraPluginConfig> {
326
458
  }
327
459
  }
328
460
 
461
+ /**
462
+ * Resolver for one embed `<type>` behind the generic
463
+ * `GET /embed/:type/:id` route. Returns the JSON body to send on
464
+ * success, or `undefined` to signal a 404 (unknown / expired id).
465
+ * `signal` aborts when the client disconnects so long-polling
466
+ * resolvers (e.g. `chart`) unblock immediately.
467
+ */
468
+ type EmbedResolver = (
469
+ req: express.Request,
470
+ id: string,
471
+ signal: AbortSignal,
472
+ ) => Promise<unknown | undefined>;
473
+
474
+ /**
475
+ * Parse the optional `?timeoutMs=<n>` query parameter from a
476
+ * `GET /embed/chart/:id` request. Accepts a positive integer up
477
+ * to 5 minutes (clamped) and rejects everything else as
478
+ * `undefined` so {@link fetchChart} falls back to its default.
479
+ * Express produces `string | string[] | undefined`; we normalize
480
+ * to the first scalar before parsing.
481
+ */
482
+ function parseTimeoutMs(raw: unknown): number | undefined {
483
+ const v = Array.isArray(raw) ? raw[0] : raw;
484
+ if (typeof v !== "string") return undefined;
485
+ const n = Number(v);
486
+ if (!Number.isFinite(n) || n <= 0) return undefined;
487
+ return Math.min(Math.floor(n), 5 * 60_000);
488
+ }
489
+
490
+ /**
491
+ * Parse the optional `?limit=<n>` query parameter from a
492
+ * `GET /embed/data/:id` request. Accepts a non-negative
493
+ * integer and lets the route clamp to `STATEMENT_ROW_CAP`;
494
+ * rejects anything else as `undefined` so the route falls back
495
+ * to the server-side cap.
496
+ */
497
+ function parseStatementLimit(raw: unknown): number | undefined {
498
+ const v = Array.isArray(raw) ? raw[0] : raw;
499
+ if (typeof v !== "string") return undefined;
500
+ const n = Number(v);
501
+ if (!Number.isFinite(n) || n < 0) return undefined;
502
+ return Math.floor(n);
503
+ }
504
+
505
+ /**
506
+ * {@link StreamFrameInterceptor} for Mastra `/stream` responses.
507
+ *
508
+ * Removes large, redundant payloads from terminal `step-finish`,
509
+ * `finish`, and `tool-result` frames before they reach the browser.
510
+ * These frames often repeat information already delivered incrementally
511
+ * via streamed text and tool events, including full tool outputs,
512
+ * accumulated responses, message history, SQL results, chart data, and
513
+ * other large result payloads.
514
+ *
515
+ * The interceptor deletes heavyweight payload properties
516
+ * (`output`, `messages`, `response`, and `result`) while preserving the
517
+ * event envelope and lifecycle metadata required by the client.
518
+ *
519
+ * Deletion is key based, so streams that do not contain these
520
+ * fields are passed through unchanged.
521
+ */
522
+ const interceptStreamFrame: StreamFrameInterceptor = (chunk) => {
523
+ if (!commonUtils.isRecord(chunk)) return true;
524
+ if (!["step-finish", "finish", "tool-result"].find((type) => type === chunk.type))
525
+ return true;
526
+ const payload = chunk.payload;
527
+ if (!commonUtils.isRecord(payload)) return true;
528
+ const trimmedPayload = commonUtils.deleteKeys(payload, [
529
+ "output",
530
+ "messages",
531
+ "response",
532
+ "result",
533
+ ]);
534
+ return trimmedPayload ? { replace: chunk } : true;
535
+ };
536
+
329
537
  export const mastra = toPlugin(MastraPlugin);
@@ -3,21 +3,24 @@
3
3
  * tool-invocation result in prior assistant messages before they
4
4
  * reach the model.
5
5
  *
6
- * Why: chartIds are only meaningful within the assistant turn that
7
- * minted them - the writer events backing them are gone after the
8
- * stream closes. When the model sees old chartIds in memory recall
9
- * (Mastra Memory persists tool results), it's tempted to type
10
- * those ids into the new turn's `[[chart:<id>]]` markers, leaving
11
- * the chat client's chart slots stuck with no matching event. This
12
- * processor removes the temptation by deleting `chartId` keys from
13
- * every assistant message's tool results before the prompt is
14
- * built. The current turn's tool results don't exist yet at
15
- * `processInput` time, so they pass through unmodified.
6
+ * Why: chartIds are turn-scoped from the model's point of view -
7
+ * each `prepare_chart` / `render_data` call mints a fresh id and
8
+ * the host UI binds it to that turn's reply. Mastra Memory
9
+ * replays prior tool results into the prompt; if old chartIds
10
+ * leak through, the model is tempted to copy them verbatim into
11
+ * the new turn's `[chart:<id>]` markers and the host UI ends up
12
+ * rendering an unrelated chart from the chart cache (or
13
+ * a 404 once the 1h TTL elapsed). This processor removes the
14
+ * temptation by deleting `chartId` keys from every assistant
15
+ * message's tool results before the prompt is built. The current
16
+ * turn's tool results don't exist yet at `processInput` time, so
17
+ * they pass through unmodified.
16
18
  *
17
19
  * The strip is recursive - any nested `chartId` field is removed,
18
- * regardless of which tool produced the result. This covers Genie's
19
- * `datasets[].chartId` and `render_data`'s top-level `chartId`
20
- * uniformly without coupling to specific tool ids.
20
+ * regardless of which tool produced the result. This covers
21
+ * `prepare_chart` / `render_data` top-level chartIds and any
22
+ * legacy `datasets[].chartId` payloads uniformly without coupling
23
+ * to specific tool ids.
21
24
  */
22
25
 
23
26
  import { logUtils } from "@dbx-tools/shared";
@@ -0,0 +1,127 @@
1
+ /**
2
+ * Databricks Statement Execution helpers for the Mastra plugin.
3
+ *
4
+ * Wraps `client.statementExecution.getStatement` with the shape
5
+ * + size + error handling the plugin's tools and the
6
+ * `/embed/data/:id` route both need:
7
+ *
8
+ * - {@link fetchStatementData}: low-level fetch that returns the
9
+ * raw `{columns, rows, rowCount}` shape used by the
10
+ * `get_statement` tool's output, the `prepare_chart` tool's
11
+ * dataset resolver, and the route's response body. Coerces
12
+ * numeric strings to numbers so downstream charts /
13
+ * aggregations don't have to.
14
+ * - {@link STATEMENT_ROW_CAP}: hard cap callers (notably the
15
+ * `/embed/data/:id` route) clamp `limit` to so a
16
+ * runaway result set can't hose a response.
17
+ * - {@link isStatementNotFoundError}: structural detector that
18
+ * normalizes the SDK's two error classes plus the loose
19
+ * `does not exist` / `not found` message shapes into a single
20
+ * boolean - lets the route map upstream 404s to a clean
21
+ * HTTP 404 without coupling to SDK error-class identity.
22
+ *
23
+ * Not Genie-specific: a Databricks `statement_id` is workspace
24
+ * scoped and lives in the Statement Execution API regardless of
25
+ * which producer (Genie, a tool, a notebook, etc.) submitted the
26
+ * query. Co-located here so consumers can fetch / cap / handle
27
+ * 404s without reaching into the Genie tool module.
28
+ */
29
+
30
+ import { ApiError, HttpError, WorkspaceClient } from "@databricks/sdk-experimental";
31
+ import type { GenieDatasetData } from "@dbx-tools/appkit-mastra-shared";
32
+ import { apiUtils } from "@dbx-tools/shared";
33
+
34
+ /**
35
+ * Hard server-side cap on rows returned by the
36
+ * `/embed/data/:id` route. Sized to keep responses small
37
+ * enough for inline tables to render snappily; the route surfaces
38
+ * a `truncated` flag whenever the upstream `rowCount` exceeds
39
+ * this so end users know they're seeing a sample.
40
+ */
41
+ export const STATEMENT_ROW_CAP = 500;
42
+
43
+ /**
44
+ * Best-effort numeric coercion for the Statement Execution API's
45
+ * all-strings cells. Leaves non-numeric strings (and explicit
46
+ * `null`s) intact; everything else flows through `Number`.
47
+ */
48
+ function coerceCell(cell: string | null): unknown {
49
+ if (cell === null) return null;
50
+ if (/^-?\d+(\.\d+)?$/.test(cell)) {
51
+ const n = Number(cell);
52
+ if (Number.isFinite(n)) return n;
53
+ }
54
+ return cell;
55
+ }
56
+
57
+ /**
58
+ * Fetch a single statement's rows via the Statement Execution API
59
+ * and reshape into the shared {@link GenieDatasetData} shape
60
+ * (column array + row records).
61
+ *
62
+ * Optional `limit` slices the returned `rows` client-side so the
63
+ * agent can scan a small sample without paging the full result
64
+ * set into context. `rowCount` always reflects the upstream total
65
+ * so callers know when the slice truncated.
66
+ *
67
+ * Exported because every consumer in the plugin (the
68
+ * `get_statement` tool, the `prepare_chart` dataset resolver, and
69
+ * the `/embed/data/:id` route) needs the exact same
70
+ * fetch + coercion pipeline so LLM-side `get_statement` output
71
+ * and UI-side `[data:<id>]` rendering stay shape-identical for
72
+ * the same `statement_id`.
73
+ */
74
+ export async function fetchStatementData(
75
+ client: WorkspaceClient,
76
+ statementId: string,
77
+ options?: { limit?: number; signal?: AbortSignal },
78
+ ): Promise<GenieDatasetData> {
79
+ const ctx = options?.signal ? apiUtils.toContext(options.signal) : undefined;
80
+ const r = await client.statementExecution.getStatement(
81
+ { statement_id: statementId },
82
+ ctx,
83
+ );
84
+ const columns = (r.manifest?.schema?.columns ?? []).map((c) => c.name ?? "");
85
+ const dataArray = (r.result?.data_array ?? []) as Array<Array<string | null>>;
86
+ const sliced =
87
+ options?.limit !== undefined && options.limit >= 0
88
+ ? dataArray.slice(0, options.limit)
89
+ : dataArray;
90
+ const rows = sliced.map((row) => {
91
+ const obj: Record<string, unknown> = {};
92
+ columns.forEach((col, i) => {
93
+ obj[col] = coerceCell(row[i] ?? null);
94
+ });
95
+ return obj;
96
+ });
97
+ return {
98
+ columns,
99
+ rows,
100
+ rowCount: r.manifest?.total_row_count ?? dataArray.length,
101
+ };
102
+ }
103
+
104
+ /**
105
+ * True when `err` looks like the Databricks SDK's "statement not
106
+ * found" error. Matches the typed {@link ApiError} 404 /
107
+ * `RESOURCE_DOES_NOT_EXIST` shape first, then falls back to the
108
+ * lower-level {@link HttpError} 404, then to a loose `does not
109
+ * exist` / `not found` message sniff for SDK shapes we haven't
110
+ * catalogued.
111
+ *
112
+ * Pulled into its own helper so callers (notably the
113
+ * `/embed/data/:id` route) stay decoupled from SDK
114
+ * error-class identity, and the conversion logic stays testable
115
+ * in isolation.
116
+ */
117
+ export function isStatementNotFoundError(err: unknown): boolean {
118
+ if (err instanceof ApiError) {
119
+ if (err.statusCode === 404) return true;
120
+ if (err.errorCode === "RESOURCE_DOES_NOT_EXIST") return true;
121
+ }
122
+ if (err instanceof HttpError && err.code === 404) return true;
123
+ if (err instanceof Error && /does not exist|not found/i.test(err.message)) {
124
+ return true;
125
+ }
126
+ return false;
127
+ }