@bike4mind/cli 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. package/README.md +1 -0
  2. package/bin/bike4mind-cli.mjs +0 -5
  3. package/dist/AgentHistoryStore-B2NEOvSW.mjs +18014 -0
  4. package/dist/{ApiClient-BOWpVvTq.mjs → ApiClient-D0fQ2FT6.mjs} +2 -2
  5. package/dist/{BubblewrapRuntime-CkL9-gnG.mjs → BubblewrapRuntime-5bLPTwEC.mjs} +2 -2
  6. package/dist/{ConfigStore-DdHHCH2t.mjs → ConfigStore-qV7NrCgZ.mjs} +1154 -1365
  7. package/dist/{ProxyManager-C1-lgzEU.mjs → ProxyManager-B0-RuR2w.mjs} +21 -4
  8. package/dist/{SandboxOrchestrator-CIegJrCj.mjs → SandboxOrchestrator-BcUv9fQ3.mjs} +46 -12
  9. package/dist/{SandboxRuntimeAdapter-ChGlxSGQ.mjs → SandboxRuntimeAdapter-BgLUVTJL.mjs} +2 -2
  10. package/dist/{SandboxRuntimeAdapter-CKelGICD.mjs → SandboxRuntimeAdapter-BmHELLuM.mjs} +1 -1
  11. package/dist/{SeatbeltRuntime-Qqt19cAN.mjs → SeatbeltRuntime-C_Y8q8Mr.mjs} +9 -1
  12. package/dist/buildAgent-C-C-VGff.mjs +2001 -0
  13. package/dist/commands/acpCommand.mjs +8 -6
  14. package/dist/commands/apiCommand.mjs +1 -1
  15. package/dist/commands/doctorCommand.mjs +1 -1
  16. package/dist/commands/envCommand.mjs +1 -1
  17. package/dist/commands/headlessCommand.mjs +26 -16
  18. package/dist/commands/mcpCommand.mjs +3 -3
  19. package/dist/commands/pluginCommand.mjs +1 -1
  20. package/dist/commands/updateCommand.mjs +1 -1
  21. package/dist/index.mjs +444 -100
  22. package/dist/{package-Bqg2LSnH.mjs → package-CGZIoxcs.mjs} +1 -1
  23. package/dist/{serve-DO9Edl5S.mjs → serve-CPXqcEZr.mjs} +2 -2
  24. package/package.json +9 -10
  25. package/dist/AgentHistoryStore-kT9eNMRO.mjs +0 -39471
  26. package/dist/ProxyManager-B1jFWL7b.mjs +0 -3
  27. package/dist/SandboxOrchestrator-BbMDgjzr.mjs +0 -3
  28. package/dist/buildAgent-B_kArQ_Y.mjs +0 -833
@@ -1,12 +1,12 @@
1
1
  #!/usr/bin/env node
2
2
  import { t as DEFAULT_SANDBOX_CONFIG } from "./types-F61_hxmG.mjs";
3
- import "crypto";
3
+ import { createHmac } from "crypto";
4
4
  import { existsSync, promises } from "fs";
5
5
  import os, { homedir } from "os";
6
6
  import path from "path";
7
7
  import { v4 } from "uuid";
8
8
  import * as path$1 from "node:path";
9
- import * as fs$2 from "node:fs";
9
+ import * as fs$1 from "node:fs";
10
10
  import * as z$2 from "zod";
11
11
  import z, { ZodError, z as z$1 } from "zod";
12
12
  import dayjs from "dayjs";
@@ -16,162 +16,9 @@ import relativeTime from "dayjs/plugin/relativeTime.js";
16
16
  import localizedFormat from "dayjs/plugin/localizedFormat.js";
17
17
  import { isAxiosError } from "axios";
18
18
  import { homedir as homedir$1 } from "node:os";
19
- import fs$1 from "fs/promises";
19
+ import fs from "fs/promises";
20
20
  process.env.APP_NAME;
21
21
  process.env.WEBSITE_URL;
22
- const SNIPPET_META_OPEN = "<!--snippet-meta";
23
- const SNIPPET_META_CLOSE = "-->";
24
- /** Index of the first non-whitespace character at or after `from`. */
25
- const skipSpaceForward = (text, from) => {
26
- let i = from;
27
- while (i < text.length && /\s/.test(text[i])) i++;
28
- return i;
29
- };
30
- /** Index just past the last non-whitespace character before `before`. */
31
- const skipSpaceBackward = (text, before) => {
32
- let i = before;
33
- while (i > 0 && /\s/.test(text[i - 1])) i--;
34
- return i;
35
- };
36
- /**
37
- * Split a prompt into its `<!--snippet-meta {...}-->` sections and the text around them.
38
- *
39
- * Scanned with indexOf rather than a regex. The regex this replaces backtracked
40
- * super-linearly on a marker whose JSON never closed, which invited a parse cap - but a
41
- * cap could not be applied safely here: the section pattern was terminated by
42
- * `(?=<!--snippet-meta|$)`, so capping moved `$` and changed section SHAPE, not just
43
- * length. Past the cap a snippet re-emitted as a `text` section, and callers that skip
44
- * snippets when collecting URLs to fetch (utils/src/llm/utils.ts) would start fetching
45
- * them. A linear scan needs no cap, so the shape is the same at every input length.
46
- */
47
- const extractSnippetMeta = (content) => {
48
- const sections = [];
49
- let cursor = 0;
50
- let search = 0;
51
- while (search < content.length) {
52
- const open = content.indexOf(SNIPPET_META_OPEN, search);
53
- if (open === -1) break;
54
- const metaStart = skipSpaceForward(content, open + 16);
55
- if (content[metaStart] !== "{") {
56
- search = open + 16;
57
- continue;
58
- }
59
- let close = content.indexOf(SNIPPET_META_CLOSE, metaStart);
60
- while (close !== -1) {
61
- const metaEnd = skipSpaceBackward(content, close);
62
- if (metaEnd > metaStart && content[metaEnd - 1] === "}") break;
63
- close = content.indexOf(SNIPPET_META_CLOSE, close + 3);
64
- }
65
- if (close === -1) break;
66
- const bodyStart = close + 3;
67
- const nextMarker = content.indexOf(SNIPPET_META_OPEN, bodyStart);
68
- const bodyEnd = nextMarker === -1 ? content.length : nextMarker;
69
- const textBefore = content.slice(cursor, open).trim();
70
- if (textBefore) sections.push({
71
- type: "text",
72
- content: textBefore
73
- });
74
- try {
75
- const meta = JSON.parse(content.slice(metaStart, skipSpaceBackward(content, close)));
76
- const snippetContent = content.slice(bodyStart, bodyEnd).trim();
77
- if (meta && snippetContent) sections.push({
78
- type: "snippet",
79
- meta,
80
- content: snippetContent
81
- });
82
- } catch (e) {
83
- console.error("Error parsing snippet meta:", e);
84
- }
85
- cursor = bodyEnd;
86
- search = bodyEnd;
87
- }
88
- const remainingText = content.slice(cursor).trim();
89
- if (remainingText) sections.push({
90
- type: "text",
91
- content: remainingText
92
- });
93
- return { sections };
94
- };
95
- function getHeader$1(headers, name) {
96
- if (!headers || typeof headers !== "object") return null;
97
- if (typeof headers.get === "function") {
98
- const val = headers.get(name);
99
- return typeof val === "string" ? val : null;
100
- }
101
- const value = headers[name] ?? headers[name.toLowerCase()];
102
- return typeof value === "string" ? value : null;
103
- }
104
- function parseNumber(value) {
105
- if (value === null || value === void 0) return null;
106
- const num = Number(value);
107
- return Number.isFinite(num) ? num : null;
108
- }
109
- /**
110
- * Parse rate limit headers from an HTTP response.
111
- *
112
- * Reads standard headers:
113
- * - `X-RateLimit-Limit` - max requests per window
114
- * - `X-RateLimit-Remaining` - requests left
115
- * - `X-RateLimit-Reset` - Unix epoch seconds when the window resets
116
- * - `Retry-After` - seconds to wait (on 429 responses), or an HTTP-date
117
- */
118
- function parseRateLimitHeaders(headers) {
119
- const limitStr = getHeader$1(headers, "X-RateLimit-Limit") ?? getHeader$1(headers, "x-ratelimit-limit");
120
- const remainingStr = getHeader$1(headers, "X-RateLimit-Remaining") ?? getHeader$1(headers, "x-ratelimit-remaining");
121
- const resetStr = getHeader$1(headers, "X-RateLimit-Reset") ?? getHeader$1(headers, "x-ratelimit-reset");
122
- const retryAfterStr = getHeader$1(headers, "Retry-After") ?? getHeader$1(headers, "retry-after");
123
- const limit = parseNumber(limitStr);
124
- const remaining = parseNumber(remainingStr);
125
- let resetAt = null;
126
- if (resetStr !== null) {
127
- const resetNum = Number(resetStr);
128
- if (Number.isFinite(resetNum)) resetAt = /* @__PURE__ */ new Date(resetNum * 1e3);
129
- else {
130
- const parsed = new Date(resetStr);
131
- if (!isNaN(parsed.getTime())) resetAt = parsed;
132
- }
133
- }
134
- let retryAfterMs = null;
135
- if (retryAfterStr !== null) {
136
- const retryNum = Number(retryAfterStr);
137
- if (Number.isFinite(retryNum)) retryAfterMs = retryNum * 1e3;
138
- else {
139
- const parsed = new Date(retryAfterStr);
140
- if (!isNaN(parsed.getTime())) retryAfterMs = Math.max(0, parsed.getTime() - Date.now());
141
- }
142
- }
143
- let usagePercent = null;
144
- if (limit !== null && limit > 0 && remaining !== null && remaining >= 0) usagePercent = Math.round((limit - remaining) / limit * 100);
145
- return {
146
- limit,
147
- remaining,
148
- resetAt,
149
- retryAfterMs,
150
- usagePercent
151
- };
152
- }
153
- function isNearLimit(info, thresholdPercent = 80) {
154
- if (info.usagePercent === null) return false;
155
- return info.usagePercent >= thresholdPercent;
156
- }
157
- /**
158
- * Build a structured log object for rate limit events.
159
- * Used by all integration clients to emit consistent log lines.
160
- */
161
- function buildRateLimitLogEntry(integration, endpoint, info, wasThrottled = false) {
162
- return {
163
- type: wasThrottled ? "RATE_LIMIT_ERROR" : "RATE_LIMIT",
164
- integration,
165
- endpoint,
166
- limit: info.limit,
167
- remaining: info.remaining,
168
- resetAt: info.resetAt?.toISOString() ?? null,
169
- retryAfterMs: info.retryAfterMs,
170
- usagePercent: info.usagePercent,
171
- wasThrottled,
172
- timestamp: (/* @__PURE__ */ new Date()).toISOString()
173
- };
174
- }
175
22
  //#endregion
176
23
  //#region ../../b4m-core/hearth/dist/index.mjs
177
24
  /**
@@ -319,23 +166,14 @@ z$1.object({
319
166
  });
320
167
  //#endregion
321
168
  //#region ../../b4m-core/common/dist/index.mjs
322
- let HttpStatus = /* @__PURE__ */ function(HttpStatus) {
323
- HttpStatus[HttpStatus["Ok"] = 200] = "Ok";
324
- HttpStatus[HttpStatus["Created"] = 201] = "Created";
325
- HttpStatus[HttpStatus["BadRequest"] = 400] = "BadRequest";
326
- HttpStatus[HttpStatus["Unauthorized"] = 401] = "Unauthorized";
327
- HttpStatus[HttpStatus["Forbidden"] = 403] = "Forbidden";
328
- HttpStatus[HttpStatus["NotFound"] = 404] = "NotFound";
329
- HttpStatus[HttpStatus["Conflict"] = 409] = "Conflict";
330
- HttpStatus[HttpStatus["UnprocessableEntity"] = 422] = "UnprocessableEntity";
331
- HttpStatus[HttpStatus["TooManyRequests"] = 429] = "TooManyRequests";
332
- HttpStatus[HttpStatus["InternalServerError"] = 500] = "InternalServerError";
333
- HttpStatus[HttpStatus["BadGateway"] = 502] = "BadGateway";
334
- return HttpStatus;
335
- }({});
336
169
  var HTTPError = class extends Error {
337
170
  statusCode;
338
171
  additionalInfo;
172
+ /**
173
+ * Set on a 5xx that reports a third party's failure the server handled correctly (a source site
174
+ * timing out), so `errorHandler` logs it at warn rather than paging as a server fault.
175
+ */
176
+ expected;
339
177
  constructor(statusCode, message, additionalInfo) {
340
178
  super(message);
341
179
  this.statusCode = statusCode;
@@ -367,47 +205,6 @@ var UnprocessableEntityError = class extends HTTPError {
367
205
  this.name = "UnprocessableEntityError";
368
206
  }
369
207
  };
370
- var BadRequestError = class extends HTTPError {
371
- additionalInfo;
372
- constructor(message, additionalInfo) {
373
- super(400, message, additionalInfo);
374
- this.additionalInfo = additionalInfo;
375
- this.name = "BadRequestError";
376
- }
377
- };
378
- var UnauthorizedError = class extends HTTPError {
379
- additionalInfo;
380
- constructor(message, additionalInfo) {
381
- super(401, message, additionalInfo);
382
- this.additionalInfo = additionalInfo;
383
- this.name = "UnauthorizedError";
384
- }
385
- };
386
- var ForbiddenError = class extends HTTPError {
387
- additionalInfo;
388
- constructor(message, additionalInfo) {
389
- super(403, message, additionalInfo);
390
- this.additionalInfo = additionalInfo;
391
- this.name = "ForbiddenError";
392
- }
393
- };
394
- var TooManyRequestsError = class extends HTTPError {
395
- additionalInfo;
396
- constructor(message, additionalInfo) {
397
- super(429, message, additionalInfo);
398
- this.additionalInfo = additionalInfo;
399
- this.name = "TooManyRequestsError";
400
- }
401
- };
402
- var CorruptedFileError = class extends HTTPError {
403
- additionalInfo;
404
- constructor(fileName, fileType, corruptionDetails, additionalInfo) {
405
- const message = `File '${fileName}' (${fileType}) appears to be corrupted${corruptionDetails ? `: ${corruptionDetails}` : ""}. Please try uploading the file again.`;
406
- super(422, message, additionalInfo);
407
- this.additionalInfo = additionalInfo;
408
- this.name = "CorruptedFileError";
409
- }
410
- };
411
208
  function isZodError(err) {
412
209
  return Boolean(err && (err instanceof ZodError || err.name === "ZodError"));
413
210
  }
@@ -800,19 +597,7 @@ z$1.object({
800
597
  content: z$1.string(),
801
598
  metadata: ArtifactMetadataSchema.optional()
802
599
  });
803
- /**
804
- * Regex sub-pattern (as a string) that matches the attribute portion of an
805
- * `<artifact ...>` opening tag. It handles:
806
- * - newlines inside the attribute list (AI sometimes wraps long tags),
807
- * - `>` characters inside double- or single-quoted attribute values.
808
- *
809
- * Exported as a string (not a compiled RegExp) so each consumer can
810
- * compose it into their own regex with the flags they need, avoiding
811
- * shared mutable `lastIndex` state.
812
- *
813
- * Usage: `new RegExp('<artifact\\s+(' + ARTIFACT_ATTRS_PATTERN + ')>...')`
814
- */
815
- const ARTIFACT_ATTRS_PATTERN = String.raw`(?:[^>"']|"[^"]*"|'[^']*')*`;
600
+ String.raw`(?:[^>"']|"[^"]*"|'[^']*')*`;
816
601
  const ClaudeArtifactMimeTypes = {
817
602
  REACT: "application/vnd.ant.react",
818
603
  HTML: "text/html",
@@ -826,45 +611,6 @@ const ClaudeArtifactMimeTypes = {
826
611
  PYTHON: "application/vnd.ant.python",
827
612
  BLOG_DRAFT: "application/vnd.b4m.blog-draft"
828
613
  };
829
- /**
830
- * Map a MIME type (or AI-provider artifact-type string) to an internal {@link ArtifactType}.
831
- *
832
- * Single source of truth - consumed by the artifact parsers (b4m-core/utils + client) and the
833
- * tool_result dedup in ChatCompletionProcess. Exact blessed-type matches first (case-insensitive,
834
- * since MIME types are), then language/format inference; returns null if unrecognized.
835
- *
836
- * This previously lived as three hand-maintained copies that drifted - e.g. the lattice
837
- * tool emits `application/vnd.b4m.lattice` but a copy matched `application/vnd.ant.lattice`,
838
- * letting lattice tool_result artifacts dodge the dedup set.
839
- */
840
- function mapMimeTypeToArtifactType(mimeType) {
841
- if (!mimeType) return null;
842
- const normalized = mimeType.toLowerCase().trim();
843
- switch (normalized) {
844
- case ClaudeArtifactMimeTypes.REACT.toLowerCase(): return "react";
845
- case ClaudeArtifactMimeTypes.HTML.toLowerCase(): return "html";
846
- case ClaudeArtifactMimeTypes.SVG.toLowerCase(): return "svg";
847
- case ClaudeArtifactMimeTypes.MERMAID.toLowerCase(): return "mermaid";
848
- case ClaudeArtifactMimeTypes.RECHARTS.toLowerCase(): return "recharts";
849
- case ClaudeArtifactMimeTypes.CHESS.toLowerCase(): return "chess";
850
- case ClaudeArtifactMimeTypes.CODE.toLowerCase(): return "code";
851
- case ClaudeArtifactMimeTypes.MARKDOWN.toLowerCase(): return "code";
852
- case ClaudeArtifactMimeTypes.LATTICE.toLowerCase(): return "lattice";
853
- case ClaudeArtifactMimeTypes.PYTHON.toLowerCase(): return "python";
854
- case ClaudeArtifactMimeTypes.BLOG_DRAFT.toLowerCase(): return "blog-draft";
855
- }
856
- if (normalized.includes("jsx") || normalized.includes("react")) return "react";
857
- if (normalized.includes("javascript") || normalized.includes("typescript")) return "code";
858
- if (normalized.includes("python") || normalized === "text/x-python") return "python";
859
- if (normalized.includes("java") || normalized.includes("c++") || normalized.includes("rust") || normalized.includes("go") || normalized.includes("ruby") || normalized.includes("php") || normalized.includes("swift") || normalized.includes("kotlin") || normalized.includes("csharp") || normalized.includes("c#")) return "code";
860
- if (normalized.includes("html") || normalized.includes("xhtml")) return "html";
861
- if (normalized.includes("svg")) return "svg";
862
- if (normalized.includes("markdown") || normalized.includes("md")) return "code";
863
- if (normalized.includes("mermaid")) return "mermaid";
864
- if (normalized.includes("recharts") || normalized.includes("chart")) return "recharts";
865
- if (normalized.includes("chess")) return "chess";
866
- return null;
867
- }
868
614
  let KnowledgeType = /* @__PURE__ */ function(KnowledgeType) {
869
615
  /**
870
616
  * A knowledge that is from a URL.
@@ -1062,9 +808,8 @@ const IMAGE_SIZE_CONSTRAINTS = {
1062
808
  },
1063
809
  /**
1064
810
  * dall-e-3 is no longer in ImageModels, but the generate path still accepts its sizes
1065
- * for callers holding a persisted one. Listed separately to record what each tier really
1066
- * accepts; isSupportedImageSize currently measures both against the union of the two
1067
- * (OPENAI_LEGACY_IMAGE_SIZES), so the split is documentation rather than enforcement.
811
+ * for callers holding a persisted one. Reached by LEGACY_DALL_E_3_MODEL_ID rather than an
812
+ * enum member; isSupportedImageSize measures each dall-e tier against its own list.
1068
813
  */
1069
814
  DALL_E_3: {
1070
815
  sizes: [
@@ -1154,6 +899,7 @@ let ChatModels = /* @__PURE__ */ function(ChatModels) {
1154
899
  ChatModels["CLAUDE_4_8_OPUS"] = "claude-opus-4-8";
1155
900
  ChatModels["CLAUDE_FABLE_5"] = "claude-fable-5";
1156
901
  ChatModels["CLAUDE_5_OPUS"] = "claude-opus-5";
902
+ ChatModels["CLAUDE_5_5_OPUS"] = "claude-opus-5-5";
1157
903
  ChatModels["JURASSIC2_ULTRA"] = "ai21.j2-ultra-v1";
1158
904
  ChatModels["JURASSIC2_MID"] = "ai21.j2-mid-v1";
1159
905
  ChatModels["GEMINI_3_5_FLASH"] = "gemini-3.5-flash";
@@ -1195,12 +941,6 @@ let ChatModels = /* @__PURE__ */ function(ChatModels) {
1195
941
  const CHAT_MODELS = Object.values(ChatModels);
1196
942
  const supportedChatModels = z$1.enum(ChatModels);
1197
943
  /**
1198
- * Every `ChatModels` Gemini entry is named `gemini...` (see the GEMINI block above) - a prefix
1199
- * check tracks that naming convention automatically as new Gemini models are added, unlike an
1200
- * explicit id list that would need updating in lockstep and could silently miss one.
1201
- */
1202
- const isGeminiModelId = (model) => model.startsWith("gemini");
1203
- /**
1204
944
  * Models that support the reasoning_effort parameter.
1205
945
  * o1-preview and o1-mini do NOT support reasoning_effort.
1206
946
  */
@@ -1221,137 +961,7 @@ const REASONING_SUPPORTED_MODELS = /* @__PURE__ */ new Set([
1221
961
  "gpt-5.6-luna",
1222
962
  "gpt-5.6-terra"
1223
963
  ]);
1224
- /**
1225
- * GPT-5-family reasoning models whose tool calling breaks on
1226
- * `/v1/chat/completions` when `reasoning_effort` is also sent. OpenAI requires
1227
- * this combination to go through `/v1/responses` instead. The failure mode
1228
- * differs by model:
1229
- * - GPT-5.4 (and -mini/-nano) hard-reject with a 400:
1230
- * "Function tools with reasoning_effort are not supported for <model> in
1231
- * /v1/chat/completions. Please use /v1/responses instead."
1232
- * - GPT-5 / -mini / -nano / 5.1 / 5.2 return 200 but silently degrade: the
1233
- * model *narrates* the tool call in its text ("Calling the tool now...")
1234
- * instead of emitting a real `tool_calls` entry, so no tool ever executes.
1235
- * This surfaced as the /opti optimizer's "Draft with AI" doing nothing on
1236
- * GPT-5: the same request on a model with `reasoning_effort`
1237
- * dropped (or on Claude) fires the tool correctly.
1238
- *
1239
- * We drop `reasoning_effort` when tools are sent for these models so tool
1240
- * calling continues to work on `/v1/chat/completions`. Dropping it only forgoes
1241
- * explicit effort control - the model still reasons at its default.
1242
- *
1243
- * NOTE: for the base GPT-5 narrator family (`RESPONSES_API_TOOL_MODELS`), the
1244
- * adapter now routes tool turns to `/v1/responses` instead - where reasoning +
1245
- * tools work together, so `reasoning_effort` is kept. This drop remains as
1246
- * defense-in-depth for the (now-unreached) chat path and covers the GPT-5.4
1247
- * family, which is NOT routed to Responses (its drop-path already works).
1248
- *
1249
- * O-series reasoning models (o1/o3/o4) are intentionally excluded: they call
1250
- * tools correctly with `reasoning_effort` on `/v1/chat/completions`.
1251
- *
1252
- * Invariant: every member here MUST also be in `REASONING_SUPPORTED_MODELS`.
1253
- * The gate in `openaiBackend.ts` short-circuits when a model doesn't support
1254
- * reasoning at all, so adding a non-reasoning model here would make the gate
1255
- * a no-op and silently leak `reasoning_effort` to the request.
1256
- */
1257
- const REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS = /* @__PURE__ */ new Set([
1258
- "gpt-5",
1259
- "gpt-5-mini",
1260
- "gpt-5-nano",
1261
- "gpt-5.1",
1262
- "gpt-5.2",
1263
- "gpt-5.4",
1264
- "gpt-5.4-mini",
1265
- "gpt-5.4-nano",
1266
- "gpt-5.6-sol",
1267
- "gpt-5.6-luna",
1268
- "gpt-5.6-terra"
1269
- ]);
1270
- /**
1271
- * GPT-5 reasoning models that silently *narrate* tool calls on
1272
- * `/v1/chat/completions` (return 200 with the call written as text instead of a
1273
- * real `tool_calls` entry, so nothing executes). The adapter
1274
- * routes these to OpenAI's `/v1/responses` API when function tools are present,
1275
- * where reasoning + tools work together and `reasoning_effort` can be kept.
1276
- *
1277
- * Deliberately excludes the GPT-5.4 family: it *hard-errors* (400) on that
1278
- * combination and is already handled by dropping `reasoning_effort` on the chat
1279
- * path (see `REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS`), so it stays on
1280
- * the working chat path to keep this routing's blast radius small. Also excludes
1281
- * `*-chat-latest` (non-reasoning) and O-series (tools work there already).
1282
- *
1283
- * Invariant: every member MUST also be in `REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS`
1284
- * so the chat path still drops `reasoning_effort` as a fallback if routing is bypassed.
1285
- */
1286
- const RESPONSES_API_TOOL_MODELS = /* @__PURE__ */ new Set([
1287
- "gpt-5",
1288
- "gpt-5-mini",
1289
- "gpt-5-nano",
1290
- "gpt-5.1",
1291
- "gpt-5.2",
1292
- "gpt-5.6-sol",
1293
- "gpt-5.6-luna",
1294
- "gpt-5.6-terra"
1295
- ]);
1296
- /**
1297
- * Models that only support temperature=1 (no custom temperature).
1298
- * Includes:
1299
- * - All reasoning models (OpenAI requires temp=1 when reasoning is active)
1300
- * - chat-latest variants that enforce this constraint
1301
- * - GPT-5.5, which rejects custom temperature even though it does not expose
1302
- * reasoning controls
1303
- */
1304
- const FIXED_TEMPERATURE_MODELS = /* @__PURE__ */ new Set([
1305
- ...Array.from(REASONING_SUPPORTED_MODELS),
1306
- "gpt-5.1-chat-latest",
1307
- "gpt-5.2-chat-latest",
1308
- "gpt-5.5"
1309
- ]);
1310
- /**
1311
- * Models that do not accept the temperature parameter at all.
1312
- * The API will reject requests that include temperature for these models.
1313
- */
1314
- const NO_TEMPERATURE_MODELS = /* @__PURE__ */ new Set([
1315
- "claude-opus-4-7",
1316
- "global.anthropic.claude-opus-4-7",
1317
- "claude-opus-4-8",
1318
- "global.anthropic.claude-opus-4-8",
1319
- "claude-sonnet-5",
1320
- "global.anthropic.claude-sonnet-5",
1321
- "claude-fable-5",
1322
- "claude-opus-5",
1323
- "kimi-k3",
1324
- "kimi-k2.7-code",
1325
- "kimi-k2.7-code-highspeed",
1326
- "kimi-k2.6",
1327
- "kimi-k2.5",
1328
- "deepseek-flash",
1329
- "deepseek-v4-pro"
1330
- ]);
1331
- /**
1332
- * Models whose safety classifiers can decline a request with `stop_reason: 'refusal'`
1333
- * (HTTP 200, empty or partial content) - Claude Fable 5's GA classifiers target research
1334
- * biology and most cybersecurity content and occasionally false-positive on benign adjacent
1335
- * work. Per Anthropic's GA guidance a refusal from these is opt-in recoverable: rather than
1336
- * surfacing a hard refusal, the backend throws so the completion loop's existing fallback
1337
- * machinery continues the request on Opus 5 (whose classifiers intervene far less often).
1338
- * A refusal from any *other* model is a genuine decline and surfaces unchanged. Keep in
1339
- * sync with the `claude-fable-5` fallback preference chain in `adminSettings/fallback.ts`.
1340
- */
1341
- const REFUSAL_FALLBACK_MODELS = /* @__PURE__ */ new Set(["claude-fable-5"]);
1342
- /**
1343
- * Bedrock-hosted Claude models that do NOT support prompt caching (`cache_control`).
1344
- * Sending `cache_control` to these models triggers a Bedrock deserialization error:
1345
- * `tools.N.cache_control: Extra inputs are not permitted`
1346
- *
1347
- * AWS Bedrock added prompt caching for Claude 3.5 Haiku and Claude 3.7 Sonnet (and later);
1348
- * the OG Claude 3 Haiku and the v1 Claude 3.5 Sonnet were not retrofitted.
1349
- *
1350
- * Keep this set narrow - default behavior is to apply caching when `cacheStrategy.enableCaching`
1351
- * is true. Add a model here only when we have concrete evidence (a Bedrock validation error)
1352
- * that it rejects `cache_control`.
1353
- */
1354
- const BEDROCK_NO_PROMPT_CACHING_MODELS = /* @__PURE__ */ new Set(["anthropic.claude-3-haiku-20240307-v1:0", "anthropic.claude-3-5-sonnet-20240620-v1:0"]);
964
+ [...Array.from(REASONING_SUPPORTED_MODELS)];
1355
965
  /**
1356
966
  * Speech to Text Models
1357
967
  *
@@ -1397,13 +1007,6 @@ z$1.enum({
1397
1007
  ...SpeechToTextModels,
1398
1008
  ...VideoModels
1399
1009
  });
1400
- /** Returns true if the model is deprecated on or before the provided date (default: now). */
1401
- const isModelDeprecated = (model, now = /* @__PURE__ */ new Date()) => {
1402
- if (!model.deprecationDate) return false;
1403
- const todayYMD = new Date(now.toISOString().slice(0, 10));
1404
- const cutoff = /* @__PURE__ */ new Date(model.deprecationDate + "T00:00:00Z");
1405
- return todayYMD.getTime() >= cutoff.getTime();
1406
- };
1407
1010
  /**
1408
1011
  * Valid status values for sub-quests.
1409
1012
  * Canonical vocabulary - the mongoose schema, zod schemas, and client all
@@ -1599,7 +1202,14 @@ const RealtimeVoiceUsageTransaction = BaseCreditTransaction.extend({
1599
1202
  });
1600
1203
  const ToolUsageTransaction = BaseCreditTransaction.extend({
1601
1204
  type: z$1.literal("tool_usage"),
1602
- model: z$1.string(),
1205
+ /**
1206
+ * The model that actually incurred the tool cost (e.g. 'gpt-image-2'), NOT the chat
1207
+ * model of the quest that ran the tool. The row is one aggregate over every charging
1208
+ * tool call in the quest, so this is set only when exactly one model charged; a quest
1209
+ * that charged on two or more models leaves it unset rather than naming one of them.
1210
+ * Per-call attribution always lives on the `feature: 'tool'` UsageEventModel rows.
1211
+ */
1212
+ model: z$1.string().optional(),
1603
1213
  questId: z$1.string(),
1604
1214
  sessionId: z$1.string()
1605
1215
  });
@@ -1685,8 +1295,8 @@ z$1.object({
1685
1295
  ownerType: z$1.enum(CreditHolderType),
1686
1296
  sessionId: z$1.string().optional(),
1687
1297
  /**
1688
- * Data lake this call is 1:1 attributable to (ingestion embeds only - a query
1689
- * embedding can span multiple lakes and is never attributed here). Unset for
1298
+ * Data lake this call is 1:1 attributable to (ingestion embeds and research-run judge calls -
1299
+ * a query embedding can span multiple lakes and is never attributed here). Unset for
1690
1300
  * every other feature/call.
1691
1301
  */
1692
1302
  dataLakeId: z$1.string().optional(),
@@ -1841,7 +1451,6 @@ const FIELD_GROUPS = [
1841
1451
  "dispatch",
1842
1452
  "availability"
1843
1453
  ];
1844
- const isFieldGroup = (value) => FIELD_GROUPS.includes(value);
1845
1454
  /** YYYY-MM-DD, the format every date-ish ModelInfo field already uses. */
1846
1455
  const CALENDAR_DATE = z$1.string().regex(/^\d{4}-\d{2}-\d{2}$/, "expected a YYYY-MM-DD calendar date");
1847
1456
  const ReasoningWrite = z$1.strictObject({
@@ -2656,6 +2265,7 @@ const DATA_LAKE_STABLE_STATUSES = [
2656
2265
  "deleted"
2657
2266
  ];
2658
2267
  DATA_LAKE_STATUSES.filter((s) => !DATA_LAKE_STABLE_STATUSES.includes(s));
2268
+ const DATA_LAKE_ORIGINS = ["curated", "connector-fed"];
2659
2269
  z$1.object({
2660
2270
  /**
2661
2271
  * The granting lake's Mongo `_id`. ALWAYS a persisted DB lake: a hardcoded/fallback lake has no
@@ -2703,6 +2313,14 @@ z$1.object({
2703
2313
  removedAt: z$1.date(),
2704
2314
  expiresAt: z$1.date()
2705
2315
  });
2316
+ /** Mirrors the read model's vocabulary deliberately (aliased, not re-declared, so the two can
2317
+ * never drift) - an audit reader learns one principal shape for both halves of the trail. */
2318
+ const LAKE_CONFIG_CHANGE_PRINCIPAL_KINDS = [
2319
+ "user",
2320
+ "agent",
2321
+ "apiKey",
2322
+ "system"
2323
+ ];
2706
2324
  /**
2707
2325
  * Every `IDataLake` field, classified as audited or not. A TOTAL map keyed by `keyof IDataLake`,
2708
2326
  * exactly like `LAKE_FIELD_VISIBILITY` in redactLakeForActor.ts and for the same reason: a list of
@@ -2731,6 +2349,7 @@ const LAKE_CONFIG_FIELD_AUDIT = {
2731
2349
  auditQueryTextEnabled: "audited",
2732
2350
  lakeMemoryEnabled: "audited",
2733
2351
  status: "audited",
2352
+ origin: "audited",
2734
2353
  createdByUserId: "audited",
2735
2354
  lastUpdatedByUserId: "excluded",
2736
2355
  fileCount: "excluded",
@@ -2739,15 +2358,58 @@ const LAKE_CONFIG_FIELD_AUDIT = {
2739
2358
  embeddingSpendMicroUsd: "excluded",
2740
2359
  lastSyncAt: "excluded",
2741
2360
  lastHealthCheckedAt: "excluded",
2361
+ lastInconsistencyScanAt: "excluded",
2742
2362
  filesDeletedAt: "excluded",
2743
2363
  filesArchivedAt: "excluded",
2744
2364
  lakeMemoryExtractionAt: "excluded",
2745
2365
  lakeMemoryCursor: "excluded",
2746
2366
  lakeMemoryPurgedAt: "excluded",
2747
2367
  inconsistencyReport: "excluded",
2748
- inconsistencyComputedAt: "excluded"
2368
+ inconsistencyComputedAt: "excluded",
2369
+ modelInconsistencyRunAt: "excluded"
2749
2370
  };
2750
2371
  [...Object.keys(LAKE_CONFIG_FIELD_AUDIT).filter((field) => LAKE_CONFIG_FIELD_AUDIT[field] === "audited")];
2372
+ z$1.object({
2373
+ dataLakeId: z$1.string(),
2374
+ /** Denormalized from the lake at offer time: a recipient route can filter by it without a join. */
2375
+ organizationId: z$1.string().nullish(),
2376
+ /** The actor who made the offer - always the eventual `grantedByUserId` of an accepted transfer. */
2377
+ offeredByUserId: z$1.string(),
2378
+ recipientUserId: z$1.string(),
2379
+ status: z$1.enum([
2380
+ "pending",
2381
+ "accepted",
2382
+ "declined",
2383
+ "cancelled",
2384
+ "expired"
2385
+ ]),
2386
+ expiresAt: z$1.date(),
2387
+ /** Set by the atomic `resolve` when the offer leaves `pending`; null while it is still open. */
2388
+ resolvedAt: z$1.date().nullish(),
2389
+ /**
2390
+ * The lake's effective owner ids AT OFFER TIME. Accept re-resolves them and refuses if they moved,
2391
+ * so an offer can never apply a transfer over another transfer (or a departure succession) that
2392
+ * happened in between. Snapshotted rather than recomputed because "stale" is exactly the case
2393
+ * where today's answer differs from the one the offer was made under.
2394
+ */
2395
+ priorOwnerUserIds: z$1.array(z$1.string()),
2396
+ offeredVia: z$1.enum([
2397
+ "grant-owner",
2398
+ "creator",
2399
+ "platform-admin",
2400
+ "org-admin"
2401
+ ]),
2402
+ /**
2403
+ * The principal to attribute the applied transfer to, resolved by the route at OFFER time (only a
2404
+ * route can tell an API key from a session). Carried so the eventual audit row keeps naming the
2405
+ * same principal the offer was made under, rather than whichever principal happens to accept.
2406
+ */
2407
+ auditPrincipal: z$1.object({
2408
+ principalKind: z$1.enum(LAKE_CONFIG_CHANGE_PRINCIPAL_KINDS),
2409
+ principalId: z$1.string(),
2410
+ onBehalfOfUserId: z$1.string().optional()
2411
+ }).optional()
2412
+ });
2751
2413
  /**
2752
2414
  * SRE Agent Trio - Shared Types
2753
2415
  *
@@ -3099,52 +2761,7 @@ let SupportedFabFileMimeTypes = /* @__PURE__ */ function(SupportedFabFileMimeTyp
3099
2761
  SupportedFabFileMimeTypes["CONF"] = "text/plain";
3100
2762
  return SupportedFabFileMimeTypes;
3101
2763
  }({});
3102
- /**
3103
- * The canonical set of MIME types the ingest pipeline can actually chunk +
3104
- * vectorize. Kept in lockstep with the `SmartChunker` switch in
3105
- * `@bike4mind/fab-pipeline`.
3106
- */
3107
- const SUPPORTED_FAB_FILE_MIME_TYPES = new Set(Object.values(SupportedFabFileMimeTypes));
3108
- /**
3109
- * Type guard: is a claimed MIME type one we actually support ingesting?
3110
- *
3111
- * Used to gate uploads so unsupported/binary files (e.g. `.exe`) are rejected
3112
- * with a clear error instead of being stored and silently "vectorized" into 0
3113
- * chunks. Node-free so it can run on both the client (upload UI) and server
3114
- * (ingest endpoints).
3115
- */
3116
- function isSupportedFabFileMimeType(mimeType) {
3117
- return !!mimeType && SUPPORTED_FAB_FILE_MIME_TYPES.has(mimeType);
3118
- }
3119
- /**
3120
- * Known MIME types emitted by the generated-audio endpoints (TTS + sound
3121
- * effects). Deliberately SEPARATE from SupportedFabFileMimeTypes: that set is
3122
- * the ingest pipeline's chunk+vectorize allowlist, and audio is intentionally
3123
- * NOT vectorizable and NOT attachable to a chat completion (no model accepts
3124
- * audio input). Kept only for extension mapping / documentation; the runtime
3125
- * guard below matches any `audio/*` so unusual provider subtypes are still
3126
- * treated as audio (i.e. still excluded from every LLM path).
3127
- */
3128
- let AudioMimeType = /* @__PURE__ */ function(AudioMimeType) {
3129
- AudioMimeType["MP3"] = "audio/mpeg";
3130
- AudioMimeType["WAV"] = "audio/wav";
3131
- AudioMimeType["OPUS"] = "audio/opus";
3132
- AudioMimeType["AAC"] = "audio/aac";
3133
- AudioMimeType["FLAC"] = "audio/flac";
3134
- AudioMimeType["PCM"] = "audio/pcm";
3135
- AudioMimeType["OGG"] = "audio/ogg";
3136
- AudioMimeType["WEBM"] = "audio/webm";
3137
- return AudioMimeType;
3138
- }({});
3139
- /**
3140
- * Is this an audio MIME type? Matches any `audio/*` (not just the known set)
3141
- * so the LLM-exclusion and vectorization-skip guards fail safe for any audio
3142
- * subtype a provider might emit.
3143
- */
3144
- function isAudioMimeType(mimeType) {
3145
- if (!mimeType) return false;
3146
- return mimeType.split(";")[0].trim().toLowerCase().startsWith("audio/");
3147
- }
2764
+ new Set(Object.values(SupportedFabFileMimeTypes));
3148
2765
  /** Reads back the classifier set by the tagged-error helpers above; `undefined` for untagged errors. */
3149
2766
  function getQuestErrorCode(error) {
3150
2767
  const code = error?.additionalInfo?.errorCode;
@@ -3223,77 +2840,12 @@ const AGENT_QUEST_MANIFEST = {
3223
2840
  }
3224
2841
  };
3225
2842
  /**
3226
- * The catalog row in force for a model and unit at a given time: newest
3227
- * effectiveFrom <= at AMONG ROWS OF THAT UNIT. Units are independent price
3228
- * streams - a newer per_minute row must never shadow the in-force per_token
3229
- * row (that would silently revert token billing to the adapter literal).
3230
- * Rows are append-only, so this is the whole time-travel story (see
3231
- * ModelPriceTypes).
3232
- */
3233
- function resolveModelPriceRow(rows, modelId, unit, at) {
3234
- let inForce;
3235
- for (const row of rows) {
3236
- if (row.modelId !== modelId || row.unit !== unit) continue;
3237
- if (row.effectiveFrom.getTime() > at.getTime()) continue;
3238
- if (!inForce || row.effectiveFrom.getTime() > inForce.effectiveFrom.getTime()) inForce = row;
3239
- }
3240
- return inForce;
3241
- }
3242
- /**
3243
- * Overlay catalog prices onto assembled ModelInfo. Only per_token rows apply
3244
- * here (they feed getTextModelCost via ModelInfo.pricing); per_minute and
3245
- * per_image rows are consumed by their own settlement paths. A model with no
3246
- * per_token row in force keeps its adapter literal - the fallback that keeps
3247
- * zero-config self-host deployments working.
3248
- */
3249
- function applyModelPriceCatalog(models, rows, at = /* @__PURE__ */ new Date()) {
3250
- if (rows.length === 0) return models;
3251
- return models.map((model) => {
3252
- if (model.type !== "text") return model;
3253
- const row = resolveModelPriceRow(rows, model.id, "per_token", at);
3254
- if (!row) return model;
3255
- const pricing = {};
3256
- for (const [threshold, tier] of Object.entries(row.pricing)) pricing[Number(threshold)] = tier;
3257
- return {
3258
- ...model,
3259
- pricing
3260
- };
3261
- });
3262
- }
3263
- /**
3264
2843
  * Input headroom the chat path holds back on top of a request's reserved output.
3265
2844
  * safeInputWindow (ChatCompletionProcess) subtracts it, so a text entry whose output
3266
2845
  * reserve leaves less than this has no room for a prompt at all.
3267
2846
  */
3268
2847
  const CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS = 1e3;
3269
2848
  /**
3270
- * Fallback context window for a media row whose catalog value is not usable (a discovery feed's
3271
- * literal 0, meaning "not applicable" rather than a real budget - see isMediaModelType). Shared by
3272
- * effectiveContextWindow (@bike4mind/utils, server) and useTokenLimits (apps/client, browser) so
3273
- * the two do not drift onto different placeholder numbers for the same "unknown" case.
3274
- */
3275
- const DEFAULT_UNKNOWN_CONTEXT_WINDOW = 2e5;
3276
- /**
3277
- * The model types this build's ModelInfo consumers narrow on. ModelRecord.type is
3278
- * wider (embedding / tts / realtime-voice), so the read path drops and counts any
3279
- * record outside this set: an old build must degrade to "I do not see the new video
3280
- * models", never to a runtime narrowing failure.
3281
- */
3282
- const MODEL_INFO_TYPES = [
3283
- "text",
3284
- "image",
3285
- "speech-to-text",
3286
- "video"
3287
- ];
3288
- const isRenderableModelType = (type) => MODEL_INFO_TYPES.includes(type);
3289
- /**
3290
- * Whether a ModelInfo type returns media (image/video) rather than tokens. Shared by
3291
- * safeInputWindow/effectiveContextWindow (@bike4mind/utils, server) and useTokenLimits
3292
- * (apps/client, browser) so "what counts as media" cannot drift between the two halves of the
3293
- * same guard - both need it, and common is the one package already safe to import from either.
3294
- */
3295
- const isMediaModelType = (type) => type === "image" || type === "video";
3296
- /**
3297
2849
  * The modalities promptMeta.model.type is allowed to record. Deliberately NARROWER than
3298
2850
  * MODEL_INFO_TYPES: 'speech-to-text' is served by its own transcription route, never by a
3299
2851
  * completion, so recording it would put a value in the field that no reader narrows on.
@@ -3305,162 +2857,6 @@ const PROMPT_META_MODEL_TYPES = [
3305
2857
  "image",
3306
2858
  "video"
3307
2859
  ];
3308
- /**
3309
- * The record -> ModelInfo adapter: one place where every ModelInfo field a
3310
- * catalog record does not carry gets its default. Each default degrades to the
3311
- * visible, recoverable behavior rather than the silent or expensive one.
3312
- *
3313
- * Pricing is never sourced here: applyModelPriceCatalog overlays the ModelPrice
3314
- * rows afterwards, and an empty map trips the [UNPRICED_MODEL] alarm on first
3315
- * billed use, which is the intended fail-loud path.
3316
- */
3317
- function toModelInfo(record) {
3318
- const retired = record.lifecycle?.status === "retired";
3319
- const disabled = record.disabled === true || record.autoDisabled === true || retired;
3320
- return {
3321
- id: record.id,
3322
- type: record.type,
3323
- name: record.name,
3324
- backend: record.backend,
3325
- contextWindow: record.contextWindow,
3326
- max_tokens: record.maxOutputTokens ?? Math.min(record.contextWindow, 4096),
3327
- ...record.maxOutputTokens === void 0 ? { maxOutputTokensDerived: true } : {},
3328
- pricing: {},
3329
- can_stream: record.canStream,
3330
- can_think: record.reasoning?.supported ?? false,
3331
- thinkingStyle: toThinkingStyle(record),
3332
- adapterFamily: record.adapterFamily,
3333
- dispatchProfile: record.dispatchProfile,
3334
- supportsVision: record.supportsVision,
3335
- supportsTools: record.supportsTools,
3336
- supportsImageVariation: record.supportsImageVariation ?? false,
3337
- supportsSafetyTolerance: record.supportsSafetyTolerance,
3338
- freeToRun: record.freeToRun,
3339
- private: record.private ?? false,
3340
- disabled,
3341
- disabledReason: record.disabledReason ?? record.autoDisabledReason ?? (retired ? "retired by the provider" : void 0),
3342
- deprecationDate: record.lifecycle?.deprecationDate,
3343
- replacedBy: record.lifecycle?.replacedBy,
3344
- trainingCutoff: record.trainingCutoff,
3345
- releaseDate: record.releaseDate,
3346
- logoFile: record.logoFile,
3347
- rank: record.rank,
3348
- description: record.description,
3349
- isSlowModel: record.isSlowModel
3350
- };
3351
- }
3352
- /**
3353
- * ModelInfo.thinkingStyle only describes the two Anthropic request shapes. Other
3354
- * reasoning styles map to undefined rather than to a wrong shape; the backends
3355
- * treat unset as their own default.
3356
- */
3357
- function toThinkingStyle(record) {
3358
- switch (record.reasoning?.style) {
3359
- case "anthropic-adaptive": return "adaptive";
3360
- case "anthropic-legacy": return "legacy";
3361
- default: return;
3362
- }
3363
- }
3364
- function fromThinkingStyle(style) {
3365
- if (style === "adaptive") return "anthropic-adaptive";
3366
- if (style === "legacy") return "anthropic-legacy";
3367
- }
3368
- /** Who makes the model, when the id namespace says so: [region.]<vendor>.<model>. */
3369
- const BEDROCK_REGION_PREFIX = /^(us|eu|apac|global)\./;
3370
- const VENDOR_BY_BACKEND = {
3371
- ["openai"]: "openai",
3372
- ["anthropic"]: "anthropic",
3373
- ["gemini"]: "google",
3374
- ["xai"]: "xai",
3375
- ["kimi"]: "moonshotai",
3376
- ["deepseek"]: "deepseek",
3377
- ["bfl"]: "black-forest-labs",
3378
- ["aws"]: "amazon",
3379
- ["voyageai"]: "voyageai",
3380
- ["ollama"]: "ollama",
3381
- ["local-image"]: "local",
3382
- ["bedrock"]: "amazon"
3383
- };
3384
- /**
3385
- * ModelInfo carries no vendor (that is one of the four disagreeing taxonomies
3386
- * this catalog replaces), so the inverse adapter derives it: the backend answers
3387
- * it for direct providers, and for Bedrock the id namespace does.
3388
- */
3389
- /**
3390
- * Bedrock id prefixes that name the same maker as a different string. AWS spells
3391
- * Kimi K2.5 `moonshotai.` and K2 Thinking `moonshot.`, so the raw prefix would
3392
- * file one vendor's two models under two vendors and split them in the admin
3393
- * dashboard. Canonicalized to the spelling the direct backend and the models.dev
3394
- * provider both use.
3395
- */
3396
- const BEDROCK_VENDOR_ALIASES = { moonshot: "moonshotai" };
3397
- function inferVendor(info) {
3398
- if (info.backend === "bedrock") {
3399
- const withoutRegion = String(info.id).replace(BEDROCK_REGION_PREFIX, "");
3400
- const dot = withoutRegion.indexOf(".");
3401
- if (dot > 0) {
3402
- const prefix = withoutRegion.slice(0, dot);
3403
- return BEDROCK_VENDOR_ALIASES[prefix] ?? prefix;
3404
- }
3405
- }
3406
- return VENDOR_BY_BACKEND[info.backend] ?? String(info.backend);
3407
- }
3408
- /**
3409
- * ModelInfo -> ModelRecord, the inverse of toModelInfo. Two callers: the
3410
- * fallback seed generator (adapter literals become seed rows) and the merge's
3411
- * base tier (a seeded model becomes a record that a catalog row can then claim
3412
- * groups of).
3413
- *
3414
- * `pricing` has no ModelInfo spelling here and is deliberately absent rather
3415
- * than guessed: catalog rows never carry it. The dispatch group round-trips
3416
- * (ModelInfo carries it since dispatch consumes it), but no feed may author it -
3417
- * a wrong value there mis-routes a request, so it stays seed- or
3418
- * operator-sourced, or comes from the seed-side DispatchResolver.
3419
- *
3420
- * Round-tripping normalizes the optional booleans toModelInfo defaults
3421
- * (can_think, private, disabled, supportsImageVariation): undefined becomes an
3422
- * explicit false. That is only observable for a model whose merged record a
3423
- * catalog row actually owns a group of.
3424
- */
3425
- function toModelRecord(info) {
3426
- return {
3427
- id: info.id,
3428
- vendor: inferVendor(info),
3429
- backend: info.backend,
3430
- type: info.type,
3431
- name: info.name,
3432
- contextWindow: info.contextWindow,
3433
- maxOutputTokens: info.maxOutputTokensDerived === true ? void 0 : info.max_tokens,
3434
- canStream: info.can_stream,
3435
- reasoning: info.can_think === void 0 && info.thinkingStyle === void 0 ? void 0 : {
3436
- supported: info.can_think === true,
3437
- style: fromThinkingStyle(info.thinkingStyle)
3438
- },
3439
- adapterFamily: info.adapterFamily,
3440
- dispatchProfile: info.dispatchProfile,
3441
- supportsVision: info.supportsVision,
3442
- supportsTools: info.supportsTools,
3443
- supportsImageVariation: info.supportsImageVariation,
3444
- supportsSafetyTolerance: info.supportsSafetyTolerance,
3445
- lifecycle: {
3446
- ...info.deprecationDate ? {
3447
- status: "deprecated",
3448
- deprecationDate: info.deprecationDate
3449
- } : { status: "active" },
3450
- ...info.replacedBy ? { replacedBy: info.replacedBy } : {}
3451
- },
3452
- description: info.description,
3453
- logoFile: info.logoFile,
3454
- rank: info.rank,
3455
- isSlowModel: info.isSlowModel,
3456
- trainingCutoff: info.trainingCutoff,
3457
- releaseDate: info.releaseDate,
3458
- private: info.private,
3459
- freeToRun: info.freeToRun,
3460
- disabled: info.disabled,
3461
- disabledReason: info.disabledReason
3462
- };
3463
- }
3464
2860
  const DEFAULT_PRICE_MARGIN = 1.2;
3465
2861
  const DEFAULT_USD_TO_CREDITS_RATE = 6e-4;
3466
2862
  /**
@@ -3508,140 +2904,6 @@ const CREDITS_PER_USD_COST = (() => {
3508
2904
  console.warn(`[pricing] Env-configured margin/rate derive ${derived} credits per USD cost; using defaults`);
3509
2905
  return Math.round(DEFAULT_PRICE_MARGIN / DEFAULT_USD_TO_CREDITS_RATE);
3510
2906
  })();
3511
- /**
3512
- * Converts a USD cost to credits, including markup, rounding UP (minimum 1).
3513
- *
3514
- * Use for reservations, eligibility checks, estimates, and display - anywhere
3515
- * a deterministic, conservative number is needed. Final settlement of variable
3516
- * usage should use usdToCreditsStochastic so users pay the exact fraction in
3517
- * expectation instead of the round-up.
3518
- *
3519
- * Examples (defaults):
3520
- * $1 USD = 2000 credits (1.2x markup at $0.0006/credit)
3521
- * $0.001 USD = 2 credits
3522
- * $0.0001 USD = 1 credit (minimum)
3523
- *
3524
- * @param rate - Credits per $1, defaults to the platform-wide CREDITS_PER_USD_COST.
3525
- * Pass an override for a provider-specific admin-tunable rate (e.g. a
3526
- * billing-sensitive external compute path priced independently of the
3527
- * platform default).
3528
- */
3529
- const usdToCredits = (usd, rate = CREDITS_PER_USD_COST) => {
3530
- return Math.max(1, Math.ceil(usd * rate));
3531
- };
3532
- /**
3533
- * Uniform draw in [0, 1) from the platform CSPRNG. Billing draws MUST be
3534
- * unpredictable: a caller who can foresee the stream could time requests to
3535
- * land on the no-charge side of every draw. Node >= 19 and all browsers
3536
- * expose globalThis.crypto; the Math.random fallback exists only so tests
3537
- * and exotic runtimes do not crash, and it warns once.
3538
- */
3539
- let warnedNonCryptoRng = false;
3540
- const cryptoUniform = () => {
3541
- const cryptoObj = globalThis.crypto;
3542
- if (cryptoObj?.getRandomValues) {
3543
- const buf = /* @__PURE__ */ new Uint32Array(1);
3544
- cryptoObj.getRandomValues(buf);
3545
- return buf[0] / 4294967296;
3546
- }
3547
- if (!warnedNonCryptoRng) {
3548
- warnedNonCryptoRng = true;
3549
- console.warn("[pricing] globalThis.crypto unavailable; billing rounding falls back to Math.random");
3550
- }
3551
- return Math.random();
3552
- };
3553
- /**
3554
- * Converts a USD cost to credits with markup using UNBIASED stochastic
3555
- * rounding: the integer part is always charged, and the fractional part
3556
- * charges one extra credit with probability equal to the fraction.
3557
- * E[charge] equals the exact fractional cost at every call size - no
3558
- * round-up overcharge, no 1-credit minimum, and no free-below-threshold
3559
- * leak. Because draws are independent and priced at exact cost, no calling
3560
- * pattern (splitting, retrying, aborting) changes expected cost.
3561
- *
3562
- * Server-side settlement only. Do NOT use for display, estimates, or
3563
- * reservations (it is non-deterministic; use usdToCredits). Zero or
3564
- * non-finite cost charges 0.
3565
- *
3566
- * @param usd - The raw provider cost in USD to convert
3567
- * @param rng - Uniform [0,1) source, injectable for tests; defaults to CSPRNG
3568
- * @param rate - Credits per $1, defaults to the platform-wide CREDITS_PER_USD_COST.
3569
- * Callers settling a provider-specific admin-tunable rate must pass the SAME
3570
- * rate value that was used at reservation time (snapshot it, don't re-read
3571
- * a live admin setting), or the settlement math mixes scales.
3572
- */
3573
- const usdToCreditsStochastic = (usd, rng = cryptoUniform, rate = CREDITS_PER_USD_COST) => {
3574
- const raw = usd * rate;
3575
- if (!Number.isFinite(raw) || raw <= 0) return 0;
3576
- const base = Math.floor(raw);
3577
- const fraction = raw - base;
3578
- return base + (rng() < fraction ? 1 : 0);
3579
- };
3580
- /**
3581
- * Output tokens the pre-flight credit hold prices when a request's max_tokens
3582
- * ceiling exceeds it.
3583
- *
3584
- * max_tokens is a *ceiling*, not a prediction: adaptive reasoning models stop at
3585
- * end_turn well short of it, so pricing the hold at the full window (128K on the
3586
- * flagship models) reserved several thousand credits per turn regardless of answer
3587
- * length. The excess was always refunded at settlement, but the hold IS the
3588
- * insufficient-funds gate, so users whose balance sat between their real cost and
3589
- * the worst case were falsely blocked.
3590
- *
3591
- * 16K covers the realistic long answer with headroom - the largest replies seen in
3592
- * practice are HTML-artifact turns at roughly 10-11K output tokens (see
3593
- * buildThinkingParams in llm-adapters/thinkingParams.ts, whose max_tokens floor was
3594
- * sized off the same measurement).
3595
- *
3596
- * UNDER-RESERVATION IS ACCEPTED, NOT PREVENTED. A turn that emits more than this
3597
- * settles as a shortfall debit at reconciliation, which is already a supported path
3598
- * (provider-basis settlement could always exceed a hold priced on the local
3599
- * estimate). The two sites clamp the shortfall differently: chat's
3600
- * computeSettlementDelta (services/llm/ChatCompletionProcess.ts) floors the debit at
3601
- * the balance snapshot taken at its OWN admission and reports the remainder as
3602
- * writtenOffCredits; cliCompletions.ts does an unclamped $inc on success and only
3603
- * logs an ALERT if it lands negative. Neither is a non-negativity guarantee: the
3604
- * chat snapshot predates any sibling turn's spend (see that function's own "best
3605
- * effort" note), and when the shortfall still fits the stale snapshot the debit
3606
- * applies in full - writtenOffCredits 0, and no BILLING_SHORTFALL_CLAMP log - so a
3607
- * concurrent holder can land negative there too. Either way the resulting balance
3608
- * fails the *next* turn's own admission gate - but that bound is per turn, not per
3609
- * holder: turns admitted concurrently are each checked against the balance at their
3610
- * own admission and settle against a snapshot that predates their siblings' spend,
3611
- * so a holder running turns in parallel can be shorted once per in-flight turn, not
3612
- * once total.
3613
- */
3614
- const PREFLIGHT_RESERVATION_OUTPUT_TOKENS = 16384;
3615
- /**
3616
- * Reservation ceiling for models that spend reasoning tokens inside their output
3617
- * budget (see reasonsWithinOutputBudget in llm-adapters/thinkingParams.ts). Those
3618
- * tokens bill as output on top of the visible answer, so the 16K figure above -
3619
- * which was measured on visible artifact size alone - under-reserves them badly.
3620
- *
3621
- * Deliberately below ADAPTIVE_THINKING_MAX_TOKENS_FLOOR (64K), which sizes the
3622
- * request's real ceiling and therefore has to cover the worst case: exceeding it
3623
- * truncates a reply mid-tag, an unrecoverable failure. A hold has no such duty -
3624
- * exceeding it settles as a shortfall debit - so it is sized for the long turn
3625
- * (roughly 3x the largest observed visible answer, leaving the rest for the trace)
3626
- * rather than the worst one, which is what keeps the gate off affordable requests.
3627
- */
3628
- const PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS = 32768;
3629
- /**
3630
- * Output-token figure to price a pre-flight credit hold at, given the max_tokens
3631
- * this request will actually send. Never raises the caller's ceiling: a request
3632
- * that asks for less than the cap holds only what it can possibly spend.
3633
- *
3634
- * Reserving only, never gating: the per-member org credit cap is still priced on
3635
- * the unshrunk ceiling at both call sites, since that check has no settlement
3636
- * counterpart to correct an under-estimate. That is a strictly larger figure than
3637
- * the hold, not a true upper bound on the turn - it prices one model round trip,
3638
- * at the uncached input rate, on the primary model - so it still under-counts a
3639
- * multi-round tool loop, a cache-write turn, or a fallback hop onto pricier pricing.
3640
- *
3641
- * @param reasonsWithinOutputBudget - reasonsWithinOutputBudget(modelInfo); passed as
3642
- * a boolean because common cannot import llm-adapters.
3643
- */
3644
- const reservationOutputTokens = (requestedMaxTokens, reasonsWithinOutputBudget = false) => Math.min(requestedMaxTokens, reasonsWithinOutputBudget ? PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS : PREFLIGHT_RESERVATION_OUTPUT_TOKENS);
3645
2907
  z.enum([
3646
2908
  "openai",
3647
2909
  "test",
@@ -3701,7 +2963,22 @@ const PromptBatchQuerySchema = z$1.object({
3701
2963
  personal: z$1.boolean().optional()
3702
2964
  });
3703
2965
  z$1.object({ queries: z$1.array(PromptBatchQuerySchema).min(1).max(32).refine((qs) => new Set(qs.map((q) => q.key)).size === qs.length, { message: "Batch query keys must be unique" }) });
3704
- z$1.object({
2966
+ /**
2967
+ * Request schema for POST /api/chat - the simplified external chat surface.
2968
+ *
2969
+ * Shared between the Next.js API handler (apps/client/pages/api/chat.ts, which
2970
+ * validates req.body with this exact object) and the OpenAPI registry
2971
+ * (b4m-core/common/src/openapi), so the published contract cannot drift from
2972
+ * what the handler actually accepts. The model is optional here and resolved
2973
+ * server-side from admin settings when omitted.
2974
+ *
2975
+ * Public-API rule: no `.catch()` / top-level `.transform()`. Both silently mutate
2976
+ * caller input (fail-quiet) and are opaque to zod-to-openapi. `historyCount` uses
2977
+ * `.default()` (fail loud on a bad value); unknown tool ids are filtered in the
2978
+ * handler (see filterKnownTools) instead of by a schema transform. This keeps the
2979
+ * schema fully OpenAPI-representable with no doc projection needed.
2980
+ */
2981
+ const SimplifiedChatRequestSchema = z$1.object({
3705
2982
  sessionId: z$1.string().nullish(),
3706
2983
  message: z$1.string(),
3707
2984
  organizationId: z$1.string().optional(),
@@ -3730,7 +3007,15 @@ z$1.object({
3730
3007
  includeSystemPrompt: z$1.boolean().optional(),
3731
3008
  systemPrompt: z$1.string().max(PROMPT_TEXT_MAX).optional().describe("System-prompt text for this request only, never persisted. Rendered as a defended block appended after every other system-prompt source, with prose instructing the model to defer to organization, session and data-lake guidance. Over the cap is a 422, never truncated.")
3732
3009
  });
3733
- z$1.object({
3010
+ /**
3011
+ * Async ACK returned on the default (wait:false) path of POST /api/chat. The
3012
+ * `type`/`errorCode` pair below is the same classifier the `wait: true` body and
3013
+ * the polled quest (`GET /api/quests/{id}`) carry, so it is modelled once here -
3014
+ * the rest of those two bodies is NOT described by this schema. The
3015
+ * handler assembles the ack body inline (apps/client/pages/api/chat.ts), so
3016
+ * this schema MUST stay in sync with that `res.json({...})` shape.
3017
+ */
3018
+ const ChatAckSchema = z$1.object({
3734
3019
  id: z$1.string(),
3735
3020
  status: z$1.string(),
3736
3021
  message_received: z$1.boolean(),
@@ -3751,7 +3036,33 @@ z$1.object({
3751
3036
  poll_url: z$1.string().optional()
3752
3037
  })
3753
3038
  });
3754
- z$1.object({
3039
+ /**
3040
+ * The quest a `wait: false` caller polls at `GET /api/quests/{id}` - the outcome
3041
+ * of the turn the ACK above only acknowledged.
3042
+ *
3043
+ * Deliberately the OUTCOME SUBSET, not the whole quest: that endpoint is a plain
3044
+ * handler rather than a contract, so this models only what decides whether the
3045
+ * turn succeeded, and a poll body carries further fields (`images`, `files`,
3046
+ * `toolPayloads`, `promptMeta`, ...). Must stay in sync with that handler's
3047
+ * `res.json` shape (apps/client/pages/api/quests/[id]/index.ts) - unlike a
3048
+ * contract-registered request/response schema, nothing validates this at
3049
+ * runtime. The "parses against the published ChatQuestPollResultSchema"
3050
+ * integration test (index.integration.test.ts) only proves the handler's
3051
+ * CURRENT response satisfies this schema - a non-strict `z.object` strips
3052
+ * unknown keys rather than rejecting them, and only `id` is required, so a
3053
+ * field the handler starts returning without a matching addition here keeps
3054
+ * that test green. The real per-field coverage lives in the sibling
3055
+ * assertions in that same test file; a shape addition still needs a schema
3056
+ * update by hand.
3057
+ *
3058
+ * A failed turn is still `status: 'done'` with the failure text in `reply`, so
3059
+ * `reply` alone cannot tell an answer from a failure - `type` and `errorCode` are
3060
+ * what separate a CLASSIFIED failure. A run recovered from a timeout with partial
3061
+ * content is not one of those: `terminalRecoveryFor` (questTimeoutRecovery.ts)
3062
+ * flips only `status` to preserve the surviving content, so it polls back as
3063
+ * `type: 'message'` even though it never finished.
3064
+ */
3065
+ const ChatQuestPollResultSchema = z$1.object({
3755
3066
  id: z$1.string(),
3756
3067
  status: z$1.enum([
3757
3068
  "stopped",
@@ -3782,7 +3093,18 @@ const ApiErrorSchema = z$1.object({
3782
3093
  */
3783
3094
  name: z$1.string().optional()
3784
3095
  });
3785
- ApiErrorSchema.extend({ errorCode: z$1.literal("insufficient_credits").optional() });
3096
+ /**
3097
+ * Error envelope for the 422 a credit-metered endpoint returns for two unrelated
3098
+ * reasons: "your body is invalid" and "you cannot afford this". `errorCode` is
3099
+ * what separates them - `insufficientCreditsError` (see insufficientCredits.ts)
3100
+ * tags the credit case, so its absence means an ordinary validation failure.
3101
+ *
3102
+ * Derived from `ApiErrorSchema` rather than re-declaring `error`/`request_id`:
3103
+ * both of those 422s are *thrown*, so errorHandler serves the body and adds
3104
+ * `name`. Extending is what keeps that documented here (and what drops it again
3105
+ * on the sunset date) instead of leaving a bespoke copy behind to drift.
3106
+ */
3107
+ const InsufficientCreditsErrorSchema = ApiErrorSchema.extend({ errorCode: z$1.literal("insufficient_credits").optional() });
3786
3108
  const supportedVoiceGenerationVendor = z.enum(["openai", "elevenlabs"]);
3787
3109
  const voiceOutputFormatSchema = z.enum([
3788
3110
  "mp3",
@@ -3793,13 +3115,12 @@ const voiceOutputFormatSchema = z.enum([
3793
3115
  "pcm"
3794
3116
  ]);
3795
3117
  const voiceResponseEncodingSchema = z.enum(["binary", "base64"]);
3796
- const TTS_MAX_INPUT_CHARS = {
3118
+ const TTS_ABSOLUTE_MAX_INPUT_CHARS = Math.max(...Object.values({
3797
3119
  openai: 4096,
3798
3120
  elevenlabs: 1e4
3799
- };
3800
- const TTS_ABSOLUTE_MAX_INPUT_CHARS = Math.max(...Object.values(TTS_MAX_INPUT_CHARS));
3121
+ }));
3801
3122
  const ttsLanguageCodeSchema = z.string().regex(/^[a-z]{2}$/, "languageCode must be a lowercase ISO 639-1 code, e.g. \"en\" or \"ja\"");
3802
- z.object({
3123
+ const ttsRequestSchema = z.object({
3803
3124
  text: z.string().min(1).max(TTS_ABSOLUTE_MAX_INPUT_CHARS),
3804
3125
  provider: supportedVoiceGenerationVendor.optional(),
3805
3126
  model: z.string().optional(),
@@ -3825,7 +3146,18 @@ const audioSaveSkippedReasonSchema = z.enum([
3825
3146
  "file_too_large",
3826
3147
  "error"
3827
3148
  ]);
3828
- z.object({
3149
+ /**
3150
+ * JSON body of `POST /api/ai/tts` when the caller asks for `encoding: 'base64'`.
3151
+ * The default `binary` encoding returns raw audio bytes instead and has no JSON
3152
+ * shape.
3153
+ *
3154
+ * The save + provider fields are all optional because the handler spreads them in
3155
+ * only when they apply: the save fields are absent when no copy was attempted
3156
+ * (`preview: true`, or the saveGeneratedAudio preference is off), and
3157
+ * `provider`/`fallbackFrom` appear only when the requested provider was
3158
+ * unavailable and another one stood in.
3159
+ */
3160
+ const ttsBase64ResponseSchema = z.object({
3829
3161
  /** Base64-encoded audio payload. */
3830
3162
  audio: z.string(),
3831
3163
  format: voiceOutputFormatSchema,
@@ -3839,7 +3171,32 @@ z.object({
3839
3171
  /** The originally requested provider that could not serve the request. */
3840
3172
  fallbackFrom: supportedVoiceGenerationVendor.optional()
3841
3173
  });
3842
- ApiErrorSchema.extend({
3174
+ /**
3175
+ * Error body for `POST /api/ai/tts`, shared by the 401, 422, 429 and 502; the 413
3176
+ * has a shape of its own (`ttsResponseTooLargeSchema`). `errorCode` is present only
3177
+ * on the conditions that carry a classifier; an ordinary validation 422 has none.
3178
+ *
3179
+ * Extends `ApiErrorSchema` because what decides whether a body carries the fields
3180
+ * errorHandler adds - `request_id`, and `name` until its 2026-12-01 sunset - is
3181
+ * whether the body was THROWN, not which status it wears, and three of these four
3182
+ * statuses are reachable both ways:
3183
+ *
3184
+ * - 401: thrown by apiKeyAuth on a rejected key; written by `auth` when no
3185
+ * credential was presented at all, and by the handler for
3186
+ * `provider_not_configured` and for an upstream credential rejection
3187
+ * (`provider_rejected`).
3188
+ * - 422: thrown by request validation and by the char-limit / format guards;
3189
+ * written by the handler for `insufficient_credits` and for an upstream 422.
3190
+ * - 429: thrown by apiKeyRateLimit; written by the handler on an upstream 429.
3191
+ * - 502: only ever written.
3192
+ *
3193
+ * So those two are genuinely optional here, and splitting this per status would be
3194
+ * wrong for the first three. The 502 does advertise both without ever sending them;
3195
+ * that is not worth a fourth error schema on one route, and `request_id` missing from
3196
+ * this handler's written bodies is a gap in the handler rather than something to
3197
+ * enshrine in a schema.
3198
+ */
3199
+ const ttsErrorResponseSchema = ApiErrorSchema.extend({
3843
3200
  provider: supportedVoiceGenerationVendor.optional(),
3844
3201
  errorCode: z.enum([
3845
3202
  "insufficient_credits",
@@ -3847,7 +3204,21 @@ ApiErrorSchema.extend({
3847
3204
  "provider_rejected"
3848
3205
  ]).optional()
3849
3206
  });
3850
- z.object({
3207
+ /**
3208
+ * 413 body: the audio was generated and billed but exceeds the serverless
3209
+ * response-size cap. When a browsable copy was saved, `fileUrl` is how the caller
3210
+ * retrieves the audio it paid for.
3211
+ *
3212
+ * Not derived from `ApiErrorSchema`: every 413 on this route is written, never
3213
+ * thrown, so errorHandler never serves one and `name` is genuinely absent rather than
3214
+ * optional - a stronger claim than `ttsErrorResponseSchema` can make for its own
3215
+ * statuses, see the note there. Two writers, and only the first matches the paragraph
3216
+ * above: the exceedsTtsResponseLimit guard, and the upstream-4xx passthrough relaying
3217
+ * a provider 413, where nothing was generated or billed and there is no `fileUrl`. An
3218
+ * oversized *request* body is a third 413 that never reaches this schema at all -
3219
+ * Next's own body parser answers it in plain text before the router runs.
3220
+ */
3221
+ const ttsResponseTooLargeSchema = z.object({
3851
3222
  error: z.string(),
3852
3223
  provider: supportedVoiceGenerationVendor,
3853
3224
  saved: z.literal(true).optional(),
@@ -3860,7 +3231,15 @@ z.enum(["openai"]);
3860
3231
  * New vendors are added here and in the `aiSoundService` factory.
3861
3232
  */
3862
3233
  const supportedSoundGenerationVendor = z.enum(["elevenlabs"]);
3863
- z.object({
3234
+ /**
3235
+ * Inbound request body for `POST /api/ai/sound-effects`.
3236
+ *
3237
+ * `durationSeconds` and `promptInfluence` bounds mirror the ElevenLabs
3238
+ * sound-generation limits (0.5-30s for the default eleven_text_to_sound_v2
3239
+ * model, prompt influence 0-1). `format` is the provider-specific output
3240
+ * encoding token (e.g. `mp3_44100_128`).
3241
+ */
3242
+ const soundEffectsRequestSchema = z.object({
3864
3243
  provider: supportedSoundGenerationVendor.default("elevenlabs"),
3865
3244
  text: z.string().min(1).max(1e3),
3866
3245
  durationSeconds: z.number().min(.5).max(30).optional(),
@@ -3878,13 +3257,23 @@ const supportedMusicGenerationVendor = z.enum(["elevenlabs"]);
3878
3257
  * change the request surface needs to accept it.
3879
3258
  */
3880
3259
  const supportedMusicModel = z.enum(["music_v1"]);
3881
- const DEFAULT_MUSIC_MODEL_ID = "music_v1";
3882
- z.object({
3260
+ /**
3261
+ * Inbound request body for `POST /api/ai/music`.
3262
+ *
3263
+ * `lengthMs` upper bound is capped below the ElevenLabs Music API ceiling to fit
3264
+ * the serving function's time budget (see MAX_MUSIC_LENGTH_MS). It carries a
3265
+ * default rather than being optional so the billed
3266
+ * length is always known up front (the reserve/settle path needs a deterministic
3267
+ * cost before generation) and the route can force that exact length on the
3268
+ * provider. `format` is the provider-specific output encoding token (e.g.
3269
+ * `mp3_44100_128`).
3270
+ */
3271
+ const musicRequestSchema = z.object({
3883
3272
  provider: supportedMusicGenerationVendor.default("elevenlabs"),
3884
3273
  prompt: z.string().min(1).max(2e3),
3885
3274
  lengthMs: z.number().int().min(3e3).max(12e4).default(1e4),
3886
3275
  forceInstrumental: z.boolean().optional(),
3887
- modelId: supportedMusicModel.default(DEFAULT_MUSIC_MODEL_ID),
3276
+ modelId: supportedMusicModel.default("music_v1"),
3888
3277
  format: z.string().optional()
3889
3278
  });
3890
3279
  VIDEO_SIZE_CONSTRAINTS.SORA.durations;
@@ -3980,32 +3369,40 @@ const AGENT_EXECUTION_STATUSES = [
3980
3369
  "aborted"
3981
3370
  ];
3982
3371
  /**
3983
- * Why a session summarization happened, stamped on `ISession.summaryTrigger`. Single source for
3984
- * the four places that each used to spell this list out: the Session zod schema
3985
- * (schemas/actions.ts), the entity type (types/entities/SessionTypes.ts), the Mongoose path
3986
- * (packages/database SessionModel) and the session.summarize event payload (apps/client
3987
- * server/utils/eventBus.ts). They drifted - the Mongoose enum said 'milestone'/'growth' for two
3988
- * values nothing produces - and a drift there is invisible at runtime, because BaseModel's
3989
- * findOneAndUpdate writes without runValidators.
3372
+ * Why a session summarization happened, stamped on `ISession.summaryTrigger`. Single source for the
3373
+ * places that each used to spell this list out: the Session zod schema (schemas/actions.ts), the
3374
+ * entity type (types/entities/SessionTypes.ts), the Mongoose path (packages/database SessionModel)
3375
+ * and the session.summarize event payload (apps/client server/utils/eventBus.ts). They drifted -
3376
+ * the Mongoose enum said 'milestone'/'growth' for two values nothing produces - and a drift there
3377
+ * is invisible on the update path, because BaseModel's findOneAndUpdate writes without
3378
+ * runValidators.
3990
3379
  *
3991
- * 'throttling' is the one member no stored document can carry: shouldSummarizeSession
3992
- * (b4m-core/services ChatCompletionFeatures) returns it as the reason it declined to summarize,
3993
- * so it never reaches a write. It stays in the union because that return value is typed as
3994
- * `ISessionDocument['summaryTrigger']`.
3380
+ * Those surfaces now name PERSISTED_SESSION_SUMMARY_TRIGGERS below, not the full union: a stored
3381
+ * field may only carry a reason a run HAPPENED. 'throttling' is the exception that forced the
3382
+ * split - shouldSummarizeSession (b4m-core/services ChatCompletionFeatures) returns it as the
3383
+ * reason it DECLINED to summarize, so it describes no run and belongs to a decision, not a
3384
+ * document. It is typed by SummarizationDecision there, not by the session field.
3995
3385
  *
3996
3386
  * 'manual' means someone asked for one notebook's summary. The admin sweep (apps/client
3997
3387
  * server/events/spider.ts) summarizes every un-summarized notebook of the admin who ran it in one
3998
3388
  * billed pass, so it stamps 'spider' instead: without that, one deliberate click and a whole sweep
3999
3389
  * are indistinguishable when someone investigates unexpected summarization spend.
4000
3390
  */
4001
- const SESSION_SUMMARY_TRIGGERS = [
3391
+ /**
3392
+ * The triggers a document may actually carry - every reason a summarization HAPPENED. The event
3393
+ * payload and createSessionParametersSchema both name this list rather than the full union below,
3394
+ * so 'throttling' cannot be published, cannot be stored, and therefore cannot reach a copy path.
3395
+ * Add a new reason-it-happened here, not to SESSION_SUMMARY_TRIGGERS, and every one of those
3396
+ * boundaries picks it up.
3397
+ */
3398
+ const PERSISTED_SESSION_SUMMARY_TRIGGERS = [
4002
3399
  "manual",
4003
3400
  "project",
4004
3401
  "earlyMilestone",
4005
3402
  "contentGrowth",
4006
- "throttling",
4007
3403
  "spider"
4008
3404
  ];
3405
+ [...PERSISTED_SESSION_SUMMARY_TRIGGERS];
4009
3406
  /**
4010
3407
  * Operator allow-list for the client-authored Mongo filter carried on a `subscribe_query` frame.
4011
3408
  *
@@ -5173,7 +4570,7 @@ const SessionCreatedAction = shareableDocumentSchema.extend({
5173
4570
  claudeConversationId: z$1.string().optional(),
5174
4571
  summary: z$1.string().optional(),
5175
4572
  summaryAt: z$1.date().optional(),
5176
- summaryTrigger: z$1.enum(SESSION_SUMMARY_TRIGGERS).optional(),
4573
+ summaryTrigger: z$1.enum(PERSISTED_SESSION_SUMMARY_TRIGGERS).optional(),
5177
4574
  deletedAt: z$1.date().optional(),
5178
4575
  tags: z$1.array(z$1.object({
5179
4576
  name: z$1.string(),
@@ -5370,7 +4767,14 @@ const PermissionRequestAction = z$1.object({
5370
4767
  executionId: z$1.string(),
5371
4768
  toolName: z$1.string(),
5372
4769
  toolInput: z$1.unknown(),
5373
- iteration: z$1.number()
4770
+ iteration: z$1.number(),
4771
+ /**
4772
+ * Provider tool_use id of the specific gated call this card is asking about.
4773
+ * The client echoes it back on `permission_response` so the server can bind
4774
+ * the answer to THIS pause rather than the latest one that happens to share
4775
+ * a tool name - see `handlePermissionResponse`'s toolCallId check.
4776
+ */
4777
+ toolCallId: z$1.string().optional()
5374
4778
  });
5375
4779
  const ChildExecutionSnapshotSchema = z$1.lazy(() => z$1.object({
5376
4780
  executionId: z$1.string(),
@@ -5392,7 +4796,8 @@ const ReconnectResultAction = z$1.object({
5392
4796
  pendingPermission: z$1.object({
5393
4797
  toolName: z$1.string(),
5394
4798
  toolInput: z$1.unknown(),
5395
- requestedAt: z$1.union([z$1.string(), z$1.date()])
4799
+ requestedAt: z$1.union([z$1.string(), z$1.date()]),
4800
+ toolCallId: z$1.string().optional()
5396
4801
  }).optional(),
5397
4802
  totalCreditsUsed: z$1.number().optional(),
5398
4803
  iterationCount: z$1.number().optional(),
@@ -5484,7 +4889,29 @@ z$1.discriminatedUnion("action", [
5484
4889
  PermissionRequestAction,
5485
4890
  ReconnectResultAction
5486
4891
  ]);
5487
- z$1.object({
4892
+ /**
4893
+ * Public wire schemas for the agent-executor (ReAct) endpoints:
4894
+ * `POST /api/v1/agent-executions` and `GET /api/v1/agent-executions/{id}`.
4895
+ *
4896
+ * These are the REST twin of the WebSocket `agent_execute` command surface
4897
+ * (apps/client/server/websocket/agentExecute.ts). Both transports funnel into the
4898
+ * same `startAgentExecution` service, but the wire shapes are deliberately separate:
4899
+ * the WS payload carries UI-only fields (routing provenance, an optimistic-bubble
4900
+ * back-reference) that must never become published API surface, and public fields are
4901
+ * snake_case per CONVENTIONS.md section 2 while the WS command is camelCase.
4902
+ *
4903
+ * Public-API rules apply here: no `.catch()`, no top-level `.transform()`.
4904
+ */
4905
+ /**
4906
+ * Request body for `POST /api/v1/agent-executions`.
4907
+ *
4908
+ * `session_id` is required rather than defaulted (unlike `POST /api/chat`, which falls
4909
+ * back to the caller's last notebook): the session is what determines which agent
4910
+ * profile the executor builds, so guessing it would silently change the run's
4911
+ * behaviour. Everything else is optional and falls back to admin defaults or the
4912
+ * agent's own orchestration profile.
4913
+ */
4914
+ const AgentExecutionStartRequestSchema = z$1.object({
5488
4915
  session_id: z$1.string().min(1),
5489
4916
  message: z$1.string().min(1),
5490
4917
  /** Falls back to the deployment's default chat model when omitted. */
@@ -5541,7 +4968,11 @@ z$1.object({
5541
4968
  */
5542
4969
  enable_artifacts: z$1.boolean().optional()
5543
4970
  });
5544
- z$1.object({
4971
+ /**
4972
+ * 202 ACK for `POST /api/v1/agent-executions`. The run is fire-and-forget: nothing is
4973
+ * streamed back over REST, so the caller polls `poll_url` until `status` is terminal.
4974
+ */
4975
+ const AgentExecutionAckSchema = z$1.object({
5545
4976
  id: z$1.string(),
5546
4977
  status: z$1.literal("pending"),
5547
4978
  session_id: z$1.string(),
@@ -5571,7 +5002,15 @@ const AgentExecutionStepSchema = z$1.object({
5571
5002
  /** Set on `action` steps: the tool the agent invoked. */
5572
5003
  tool_name: z$1.string().optional()
5573
5004
  });
5574
- z$1.object({
5005
+ /**
5006
+ * Poll response for `GET /api/v1/agent-executions/{id}`.
5007
+ *
5008
+ * `steps` is the live trace: it grows while the run is in flight (read from the
5009
+ * checkpoint) and freezes at the final one. `answer` is null until the run reaches a
5010
+ * terminal status, and stays null on `failed` / `aborted` - where `error` carries the
5011
+ * reason instead.
5012
+ */
5013
+ const AgentExecutionStatusResponseSchema = z$1.object({
5575
5014
  id: z$1.string(),
5576
5015
  status: z$1.enum([
5577
5016
  "pending",
@@ -5603,7 +5042,8 @@ z$1.object({
5603
5042
  created_at: z$1.string(),
5604
5043
  updated_at: z$1.string()
5605
5044
  });
5606
- z$1.object({ id: z$1.string().min(1) });
5045
+ /** Path parameter for `GET /api/v1/agent-executions/{id}`. */
5046
+ const AgentExecutionIdParamSchema = z$1.object({ id: z$1.string().min(1) });
5607
5047
  /**
5608
5048
  * Tool schema matching ICompletionOptionTools.toolSchema. The Zod surface only
5609
5049
  * covers wire-format fields (toolFn is server-side). Replaces the historical
@@ -5653,7 +5093,17 @@ const CompletionMessageSchema = z$1.object({
5653
5093
  content: z$1.union([z$1.string(), z$1.array(z$1.any())]),
5654
5094
  cache: z$1.boolean().optional()
5655
5095
  });
5656
- z$1.object({
5096
+ /**
5097
+ * Schema for CLI LLM completion requests
5098
+ * Shared between Next.js API route (dev) and Lambda function (production)
5099
+ *
5100
+ * `response_format`, `stream`, `tools`, `temperature`, and `max_tokens` are all
5101
+ * accepted at the top level (OpenAI-compatible, matching how every major LLM
5102
+ * SDK shapes a completion request) AND nested under `options` (legacy shape).
5103
+ * Use `normalizeCompletionRequest()` to collapse both surfaces into the
5104
+ * canonical `options.<field>` location before downstream consumption.
5105
+ */
5106
+ const CompletionRequestSchema = z$1.object({
5657
5107
  model: z$1.string(),
5658
5108
  messages: z$1.array(CompletionMessageSchema),
5659
5109
  response_format: ResponseFormatSchema.optional(),
@@ -5722,12 +5172,20 @@ const CompletionSseErrorEventSchema = z$1.object({
5722
5172
  requestId: z$1.string().optional(),
5723
5173
  code: z$1.enum(QUEST_ERROR_CODES).optional()
5724
5174
  });
5725
- z$1.union([
5175
+ /** One `data:` event in the `text/event-stream` completions response. */
5176
+ const CompletionStreamEventSchema = z$1.union([
5726
5177
  CompletionMetaEventSchema,
5727
5178
  CompletionContentEventSchema,
5728
5179
  CompletionSseErrorEventSchema
5729
5180
  ]);
5730
- z$1.object({
5181
+ /**
5182
+ * Server-side tool execution schemas for POST /api/ai/v1/tools.
5183
+ *
5184
+ * Plain Zod (no `.openapi()`) so any runtime can import them; the OpenAPI layer
5185
+ * annotates them via the contract. The tool-name enum MUST stay in sync with
5186
+ * SUPPORTED_TOOLS in apps/client/server/cli/toolsHandler.shared.ts.
5187
+ */
5188
+ const ToolExecutionRequestSchema = z$1.object({
5731
5189
  toolName: z$1.enum([
5732
5190
  "weather_info",
5733
5191
  "web_search",
@@ -6446,13 +5904,6 @@ const ImageOutputFormatSchema = z$1.enum([
6446
5904
  "webp"
6447
5905
  ]);
6448
5906
  /**
6449
- * Degrade a shared output-format setting to what BFL and Gemini accept, so selecting
6450
- * webp for gpt-image cannot fail an unrelated render after a model switch.
6451
- */
6452
- function toNonWebpOutputFormat(format) {
6453
- return format === "webp" ? "png" : format;
6454
- }
6455
- /**
6456
5907
  * Maps legacy/removed image model IDs to their current replacements.
6457
5908
  * Prevents Zod validation failures when clients send stale persisted model names.
6458
5909
  *
@@ -6611,7 +6062,13 @@ const SessionTagSchema = z$1.object({
6611
6062
  name: z$1.string(),
6612
6063
  strength: z$1.number()
6613
6064
  });
6614
- z$1.object({
6065
+ /**
6066
+ * Request schema for PUT /api/sessions/{id}. This is the exact field allowlist
6067
+ * sessionService.updateSession enforces (b4m-core/services/src/sessionService/update.ts,
6068
+ * which extends this schema with `id`) - shared so the public contract can never
6069
+ * document a field the service silently drops, or vice versa.
6070
+ */
6071
+ const SessionUpdateRequestSchema = z$1.object({
6615
6072
  name: z$1.string().min(1).optional(),
6616
6073
  knowledgeIds: z$1.array(z$1.string()).optional(),
6617
6074
  artifactIds: z$1.array(z$1.string()).optional(),
@@ -6621,8 +6078,14 @@ z$1.object({
6621
6078
  lakeScope: z$1.array(z$1.string()).nullable().optional().describe("The data lakes this session grounds on, as lake tags (the `datalakeTag` of each lake from GET /api/data-lakes). Send a list to ground only on those lakes, `[]` to ground on no lake at all, or `null` to clear the choice so retrieval falls back to every lake you can reach. Omit to leave the current choice unchanged. Tags naming a lake you cannot reach are ignored at retrieval time rather than rejected here. Narrowing the scope does not by itself turn retrieval on: pair it with `forceKnowledgeRetrieval: true` for a session that is not already grounded. Conversely `[]` leaves a grounded session nothing to retrieve from, so its forced retrieval is skipped rather than run against every lake."),
6622
6079
  propagateToProjects: z$1.boolean().optional().describe("Defaults to true when omitted. When knowledgeIds grows, the newly-added file ids are also appended to every project that contains this session, granting every member of that project access to those files. This propagation is append-only and cannot be undone through the UI - pass false if newly-attached files should not be shared with the project.")
6623
6080
  });
6624
- z$1.object({ id: z$1.string().min(1) });
6625
- z$1.object({
6081
+ /** Path parameter for session-scoped endpoints, e.g. GET/PUT /api/sessions/{id}. */
6082
+ const SessionIdParamSchema = z$1.object({ id: z$1.string().min(1) });
6083
+ /**
6084
+ * Practical response subset for PUT /api/sessions/{id} - the fields a caller needs to
6085
+ * confirm an update took effect. ISession (types/entities/SessionTypes.ts) carries many
6086
+ * more server-internal fields not documented as public API surface here.
6087
+ */
6088
+ const SessionResponseSchema = z$1.object({
6626
6089
  id: z$1.string(),
6627
6090
  name: z$1.string(),
6628
6091
  userId: z$1.string(),
@@ -6751,15 +6214,6 @@ const CHUNK_STALL_REASONS = [
6751
6214
  "unchunkedPaused"
6752
6215
  ];
6753
6216
  /**
6754
- * Whether a file is stalled by the convergence kill switch, by any arm. THE predicate every
6755
- * reader uses, so adding a stall reason reaches health, convergence and retrieval without separate
6756
- * comparisons drifting apart. Also the in-memory mirror of a Mongo
6757
- * `chunkStallReason: { $in: [...CHUNK_STALL_REASONS] }`.
6758
- */
6759
- function isChunkStalled(reason) {
6760
- return CHUNK_STALL_REASONS.includes(reason);
6761
- }
6762
- /**
6763
6217
  * Which reasons leave the file with NO passages, as opposed to passages with no vectors. A `Record`
6764
6218
  * over every reason rather than a hand-written subset array: a new stall reason then cannot compile
6765
6219
  * until it is classified, where a member missing from a literal array would just make a health count
@@ -6786,76 +6240,7 @@ const CHUNK_STALL_NOTICES = {
6786
6240
  };
6787
6241
  CHUNK_STALL_NOTICES.vectorizePaused;
6788
6242
  CHUNK_STALL_NOTICES.rechunkPaused;
6789
- /**
6790
- * TRANSITIONAL, and the ONE stall predicate every RETRIEVAL path must use until #2016's migration
6791
- * has run in every environment. Reads the new field, then falls back to the legacy prose that the
6792
- * pre-migration rows still carry in `notes`.
6793
- *
6794
- * It exists for the FORWARD window only: `migratorInvocation` is a `dependsOn` of the web stack
6795
- * only (infra/web.ts); the queue stack has none, so the executor can serve forced retrieval and
6796
- * `knowledge_base_search` while rows still carry the marker in `notes` and no `chunkStallReason`. A
6797
- * row stalled by the chunk arm then reads as a plain unindexed file: `isRetrievalExcluded` drops it
6798
- * upstream of the withhold on a vectorizedOnly lake, and `partitionByIndexAvailability` calls it
6799
- * servable everywhere else. The turn answers around a passage-less file and reports FULL coverage -
6800
- * the silent degradation this whole path exists to prevent.
6801
- *
6802
- * A code ROLLBACK is the mirror image and this arm CANNOT cover it: the rows are already migrated
6803
- * (`chunkStallReason` set, `notes` unset) and the code restored is pre-#2016, which does not contain
6804
- * this function. Nothing reverts the data on its own either - `migratorInvocation` only ever runs
6805
- * `up` and `migrate down` is a manual CLI step - so `migrate down` is a REQUIRED step of any
6806
- * rollback past #2016, not an optional tidy-up. What this arm does buy is that `down()` is safe to
6807
- * run FIRST: whichever stack is still new keeps honoring the prose it restores, so a staggered
6808
- * rollback has no window where a restored marker is invisible. `down()` is a PARTIAL restore
6809
- * though - it skips a row whose owner typed a note after `up()`, and that row grades as unstalled
6810
- * on both stacks once the field is dropped. See its own comment.
6811
- *
6812
- * Deliberately NOT used by the grading/health/UI readers: they are gated behind the web stack, and
6813
- * a legacy row there renders the notice line AND the identical text as the owner's note.
6814
- *
6815
- * Mirrored in Mongo by `buildFabFileSearchQuery`'s `vectorizedOnly` exemption. Delete the legacy arm
6816
- * from both together, one release after the migration has landed everywhere.
6817
- *
6818
- * Pinned to the two reasons the migration backfilled rather than every notice: `unchunkedPaused`
6819
- * postdates it, so no row carries its prose, and including it would read an owner who happens to type
6820
- * that sentence into `notes` as stalled.
6821
- */
6822
- const LEGACY_CHUNK_STALL_NOTES = [CHUNK_STALL_NOTICES.vectorizePaused, CHUNK_STALL_NOTICES.rechunkPaused];
6823
- function isChunkStalledFile(file) {
6824
- return isChunkStalled(file.chunkStallReason) || LEGACY_CHUNK_STALL_NOTES.includes(file.notes ?? "");
6825
- }
6826
- /**
6827
- * `FabFile.chunkRebuildRequestedAt`: stamped by `resetChunkStateByIds` in the SAME write that
6828
- * clears a file's chunk rollups, so "this file's passages are being rebuilt" can never be lost the
6829
- * way the pair of steps that creates the state can be. The reset and the queue send are two
6830
- * operations - kill the producer between them, or lose the consumer's marker write, and the file
6831
- * sits at `chunkCount: 0` with `error: null` and no stall reason, a shape indistinguishable from an
6832
- * image or a still-uploading row. It then drops out of lake health's denominator, out of the
6833
- * convergence plan and out of the retrieval withhold at the same moment: every rollup says its
6834
- * passages are gone, and nothing reports it.
6835
- *
6836
- * Deliberately NOT the `rechunkPaused` stall reason pre-written by the producer, which is the obvious
6837
- * fix and the wrong one: that marker means "halted, needs an administrator", so a file awaiting an
6838
- * ORDINARY rebuild would read to every reader as permanently paused for the whole rebuild - search
6839
- * would tell readers it does not return on its own, health would hard-fail P3, and "Rebuild
6840
- * passages" would offer to repair a file that is already repairing. A flag that cries wolf on the
6841
- * normal path is worse than the rare window it closes.
6842
- *
6843
- * So the two facts are distinct states, and the consumer UPGRADES one to the other: pending means
6844
- * "in flight, returns on its own", the paused note means "halted, needs intervention". A LOST
6845
- * upgrade therefore degrades to mislabelled-but-visible rather than invisible, which is the trade
6846
- * this field exists to make - invisibility is the real harm, labelling is secondary.
6847
- *
6848
- * A dedicated field on purpose, and the precedent #2016 followed for the other two machine-written
6849
- * facts: while they all shared `notes` every writer of that field clobbered the others, including
6850
- * the user's own note.
6851
- *
6852
- * Cleared by `commitFabFileChunks` (the rebuild landed) and by the chunk handler's pause write (the
6853
- * rebuild was halted instead). A file carrying `error` is settled regardless - see
6854
- * `isMemberIndexingInFlight`, which is where the precedence between these three lives.
6855
- */
6856
- function isChunkRebuildPending(requestedAt) {
6857
- return requestedAt !== null && requestedAt !== void 0 && requestedAt !== "";
6858
- }
6243
+ CHUNK_STALL_NOTICES.vectorizePaused, CHUNK_STALL_NOTICES.rechunkPaused;
6859
6244
  /** Ceiling so "adjustable" cannot mean "unbounded" in either direction. */
6860
6245
  const LAKE_ACCESS_AUDIT_RETENTION_MAX_DAYS = 2555;
6861
6246
  /**
@@ -6949,47 +6334,8 @@ function defaultEmbeddingModelForEnv() {
6949
6334
  const selfHost = process.env.B4M_SELF_HOST === "true";
6950
6335
  const hasOllama = !!process.env.OLLAMA_BASE_URL?.trim();
6951
6336
  const hasCloudEmbeddingKey = !isPlaceholderApiKey(process.env.OPENAI_API_KEY) || !isPlaceholderApiKey(process.env.VOYAGE_API_KEY);
6952
- if (selfHost && hasOllama && !hasCloudEmbeddingKey) return "qwen3-embedding:0.6b";
6953
- return "text-embedding-3-small";
6954
- }
6955
- /**
6956
- * True when this deployment can embed with no provider API key at all: a cloud stage reaches
6957
- * Bedrock through its task/execution role's AWS credentials.
6958
- *
6959
- * Requires POSITIVE evidence of an execution role rather than merely "not self-host". A plain
6960
- * `next dev` session and a CI job both leave B4M_SELF_HOST unset while holding no AWS credentials
6961
- * at all, so an absence test would send them to the Bedrock SDK for an opaque `CredentialsProvider
6962
- * Error` in place of the actionable OPENAI_KEY_MISSING_MESSAGE naming the key to set - the same
6963
- * actionable-to-opaque trade this fallback exists to avoid, just in a different keyless place.
6964
- *
6965
- * BOTH runtimes must be covered, and they carry different markers. The Lambdas (vectorize
6966
- * subscriber, the crons, the Next API routes) get AWS_LAMBDA_FUNCTION_NAME; ChatCompletion is a
6967
- * Fargate service (infra/chatCompletion.ts) and gets the ECS task-role URI instead. Since
6968
- * knowledgeBaseSearch runs inside that container, keying on the Lambda marker alone would leave
6969
- * chat knowledge-base search failing on exactly the keyless stages this fallback is for.
6970
- *
6971
- * SST_RESOURCE_App is the third arm and the one this repo can prove: SST sets it on anything it
6972
- * links, Lambda and Service alike (infra/chatCompletion.ts:129 and infra/agentExecutor.ts:54 both
6973
- * note that linking alone exposes SST_RESOURCE_*). It covers the Fargate task whether or not the
6974
- * ECS credential URI is present, and it is absent from a plain `next dev` and from CI, which is
6975
- * the case that matters. `sst dev` does set it - correctly, since that session runs against real
6976
- * AWS credentials.
6977
- *
6978
- * Self-host is excluded outright because it has no such role - its keyless path is the local
6979
- * Ollama embedder (`isLocalEmbedderAvailable` in toolAvailability.ts), not Bedrock.
6980
- *
6981
- * Answers "is Bedrock reachable here", NOT "should we use it" - a keyed stage is keyless-capable
6982
- * too, so this must only ever be asked ALONGSIDE a resolved credential table that came back empty.
6983
- * `resolveEmbeddingWithKeylessFallback` is where the two questions are paired for callers free to
6984
- * choose the model, and is what such a caller should use instead of asking this directly. The
6985
- * direct callers are the ones that additionally need the answer BEFORE resolving, to decide
6986
- * policy: toolAvailability reports whether embedding-backed tools are usable at all, and
6987
- * data-lakes/semantic-search decides whether a substitution is permitted for this request before
6988
- * it knows whether one is needed. Both still pair it with the table.
6989
- */
6990
- function hasKeylessCloudEmbedder() {
6991
- if (process.env.B4M_SELF_HOST === "true") return false;
6992
- return !!(process.env.AWS_LAMBDA_FUNCTION_NAME || process.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI || process.env.AWS_CONTAINER_CREDENTIALS_FULL_URI || process.env.SST_RESOURCE_App);
6337
+ if (selfHost && hasOllama && !hasCloudEmbeddingKey) return "qwen3-embedding:0.6b";
6338
+ return "text-embedding-3-small";
6993
6339
  }
6994
6340
  z$1.union([
6995
6341
  z$1.enum(OpenAIEmbeddingModel),
@@ -7182,20 +6528,6 @@ Call \`search_knowledge_base\` BEFORE answering when that library would settle t
7182
6528
  Do not search when the answer is already in front of you or out of scope: general knowledge (definitions, mathematics, established theory, public facts); anything answerable from this conversation, from an attached document, or from content already retrieved for you this turn; or a request to transform, summarize or reformat text the user has just supplied. If the library has already been searched on this turn, do not search it again for the same question - a repeat spends a round trip to return the same passages.
7183
6529
 
7184
6530
  When a search does not turn up what was asked for, say so plainly rather than filling the gap from training data, and never imply an answer came from the user's documents when it did not.`;
7185
- /**
7186
- * Default text for the formatting system message. Runtime fallback used by
7187
- * `includeHardcodedSystemMessage` (b4m-core/utils/src/llm/utils.ts) when the `FormatPromptTemplate`
7188
- * admin setting is blank; that setting's own default is intentionally '' - keep this the sole home.
7189
- *
7190
- * Deliberately scoped to formatting ONLY. The previous wording ("Adhere to specific formatting
7191
- * requests...") read as a general compliance instruction and bled into WHETHER to answer: as the
7192
- * only system content it roughly halved refusal quality. The opening clause is the fix - it fences
7193
- * this message off from the answer/abstain decision. Injected only when `UseFormatPrompt` is on.
7194
- *
7195
- * NOTE: a stored settings row pins its own wording, so changing this default does not reach an
7196
- * existing deployment that has already saved a value - the row must be edited in admin settings too.
7197
- */
7198
- const FORMAT_PROMPT_TEMPLATE = `Formatting only - nothing here decides whether or how fully to answer. Format replies to maintain the integrity of the requested style; default to markdown for text. Preserve proper structure for poems, songs, or haikus. When the user specifies an output format (e.g. TypeScript), use that format for the parts you do answer.`;
7199
6531
  z$1.enum([
7200
6532
  "openaiDemoKey",
7201
6533
  "anthropicDemoKey",
@@ -7239,13 +6571,16 @@ z$1.enum([
7239
6571
  "EnableDataLakeSlackAdd",
7240
6572
  "EnableDataLakeGroundingMode",
7241
6573
  "EnableLakeMemory",
6574
+ "EnableLakeModelInconsistencyDetection",
7242
6575
  "EnableDataLakeVectorSearch",
7243
6576
  "EnableRetrievalSupersessionCollapse",
7244
6577
  "PauseLakeConvergence",
7245
6578
  "LakeConvergenceBulkChangeSharePct",
7246
6579
  "EnforceLakeReadGrants",
7247
6580
  "EnableDataLakeDrivePoll",
6581
+ "EnableDataLakeGitHub",
7248
6582
  "EnforceLakeAdmission",
6583
+ "EnforceLakeOriginOnIngest",
7249
6584
  "EnableBriefcase",
7250
6585
  "EnableBriefcaseDefault",
7251
6586
  "EnableImageTemplates",
@@ -7451,7 +6786,13 @@ const IntentClassifierConfigSchema = z$1.object({
7451
6786
  fallbackModels: z$1.array(z$1.string()).default(["gemini-2.5-flash-lite", "gpt-5.4-nano"])
7452
6787
  });
7453
6788
  const OrchestrationDefaultsSchema = z$1.object({
7454
- /** Tool names the synthetic profile is allowed to invoke. */
6789
+ /**
6790
+ * Tool names the synthetic profile is allowed to invoke. A DEFAULT toolbelt, not a gate:
6791
+ * an agentless chat dispatch ships the user's ambient Smart Tools and the executor UNIONS
6792
+ * them onto this list (`pickEffectiveEnabledTools`), so narrowing this narrows what the
6793
+ * agent brings of its own rather than capping what the user may select. `deniedTools` below
6794
+ * is the gate.
6795
+ */
7455
6796
  allowedTools: z$1.array(z$1.string()).default([
7456
6797
  "web_search",
7457
6798
  "retrieve_knowledge_content",
@@ -9212,6 +8553,16 @@ const settingsMap = {
9212
8553
  order: 91,
9213
8554
  dependsOn: "EnableDataLakes"
9214
8555
  }),
8556
+ EnableLakeModelInconsistencyDetection: makeBooleanSetting({
8557
+ key: "EnableLakeModelInconsistencyDetection",
8558
+ name: "Data Lakes: Model-driven contradiction pass",
8559
+ defaultValue: false,
8560
+ description: "Gate for the model-driven reading pass (#3057) that finds cross-document contradictions the lexical pattern rules cannot - two documents stating incompatible things in ordinary prose. Off by default: unlike the free lexical pass, this reads corpus content through an LLM, so it costs real money per run. Findings land in the same durable findings collection (detector: 'model') as the lexical pass, triggered the same way (POST /api/data-lakes/:id/inconsistencies?detector=model), gated separately here and rate-limited far lower per caller. Detect only - see the guardrail on corpusInconsistency.ts.",
8561
+ category: "Experimental",
8562
+ group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8563
+ order: 98,
8564
+ dependsOn: "EnableDataLakes"
8565
+ }),
9215
8566
  EnableDataLakeVectorSearch: makeBooleanSetting({
9216
8567
  key: "EnableDataLakeVectorSearch",
9217
8568
  name: "Data Lakes: Use Atlas $vectorSearch",
@@ -9283,6 +8634,16 @@ const settingsMap = {
9283
8634
  order: 95,
9284
8635
  dependsOn: "EnableDataLakes"
9285
8636
  }),
8637
+ EnableDataLakeGitHub: makeBooleanSetting({
8638
+ key: "EnableDataLakeGitHub",
8639
+ name: "Data Lakes: GitHub repository source",
8640
+ defaultValue: false,
8641
+ description: "Server-side gate for connecting a GitHub repository to a data lake through the read-only GitHub App (contents:read + metadata:read on the one selected repository). Off by default while the connect, ingest and purge pieces land dark; every GitHub lake route answers 403 until it is on.",
8642
+ category: "Experimental",
8643
+ group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8644
+ order: 96,
8645
+ dependsOn: "EnableDataLakes"
8646
+ }),
9286
8647
  EnforceLakeAdmission: makeBooleanSetting({
9287
8648
  key: "EnforceLakeAdmission",
9288
8649
  name: "Data Lakes: Enforce the admission contract",
@@ -9298,6 +8659,21 @@ const settingsMap = {
9298
8659
  "lake"
9299
8660
  ] }
9300
8661
  }),
8662
+ EnforceLakeOriginOnIngest: makeBooleanSetting({
8663
+ key: "EnforceLakeOriginOnIngest",
8664
+ name: "Data Lakes: Enforce curated-lake origin on ingest",
8665
+ defaultValue: true,
8666
+ description: "ON by default: unattended ingest (the Drive folder sync) refuses to add content to a lake whose owner declared it curated. OFF makes the refusal advisory and lets the write through. Unlike the admission contract this ships ON, because it refuses on an explicit owner declaration rather than a heuristic, and because the origin backfill marks every lake that currently has a connector as connector-fed - so at rollout this refuses nothing that exists. The lake rung is the one that matters; the org and owner rungs disable it across every lake in that scope at once. A flip is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min elsewhere.",
8667
+ category: "Experimental",
8668
+ group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8669
+ order: 97,
8670
+ dependsOn: "EnableDataLakes",
8671
+ scope: { settableAt: [
8672
+ "organization",
8673
+ "owner",
8674
+ "lake"
8675
+ ] }
8676
+ }),
9301
8677
  EnableBriefcase: makeBooleanSetting({
9302
8678
  key: "EnableBriefcase",
9303
8679
  name: "Enable Briefcase",
@@ -12074,7 +11450,28 @@ const CitableSourceSchema = z$1.object({
12074
11450
  practiceAreas: z$1.array(z$1.string()).optional(),
12075
11451
  chunkId: z$1.string().optional(),
12076
11452
  relevanceScore: z$1.number().optional(),
12077
- fullContext: z$1.string().optional()
11453
+ fullContext: z$1.string().optional(),
11454
+ /** web_search's own thumbnail/image cluster for this source, gated on `withImages`. */
11455
+ thumbnail: z$1.string().optional(),
11456
+ images: z$1.array(z$1.string()).optional(),
11457
+ /**
11458
+ * Ids of the other cited sources this one provably disagrees with (#3041). Declared rather
11459
+ * than left to the loose object, for the same reason chunkId/fullContext are: a writer that
11460
+ * stamps the wrong shape should fail here, not render a badge that silently names nobody.
11461
+ */
11462
+ conflictsWith: z$1.array(z$1.string()).optional(),
11463
+ /** web_search's provider-located place (WebSearchPlace), the only source of map coordinates. */
11464
+ place: z$1.object({
11465
+ id: z$1.string(),
11466
+ name: z$1.string(),
11467
+ lat: z$1.number(),
11468
+ lng: z$1.number(),
11469
+ rating: z$1.number().optional(),
11470
+ reviews: z$1.number().optional(),
11471
+ category: z$1.string().optional(),
11472
+ address: z$1.string().optional(),
11473
+ thumbnail: z$1.string().optional()
11474
+ }).optional()
12078
11475
  }).optional()
12079
11476
  });
12080
11477
  /**
@@ -12461,7 +11858,44 @@ const RetrievalSummarySchema = z$1.object({
12461
11858
  * absence is weaker evidence than presence. Date-bound any rollup: turns predating this field
12462
11859
  * carry nothing, and no backfill is possible - a past turn's grant rows have moved on.
12463
11860
  */
12464
- grantedLakeIdsUsed: z$1.array(z$1.string()).optional()
11861
+ grantedLakeIdsUsed: z$1.array(z$1.string()).optional(),
11862
+ /**
11863
+ * How many lakes were excluded from this turn's scope because the caller lacks the access to
11864
+ * search them, and why (#3055). Resolved at the seed alongside `lakeScope`, from a dedicated
11865
+ * count-only query (see excludedByAccessCount on getDynamicDataLakeAccess - NOT derived from
11866
+ * the candidate set `lakeScope` comes from, which already has the gate enforced datastore-side
11867
+ * and so cannot see this population).
11868
+ *
11869
+ * ABSENT MEANS NOT RECORDED, never "nothing was excluded" - a turn with nothing excluded records
11870
+ * `count: 0` explicitly. Three distinct causes collapse into this one absent state and are not
11871
+ * distinguishable from it: a turn predating this field, a turn whose `retrieval` was written
11872
+ * only by a tool arm rather than by the seed, and the count-only query itself failing or not
11873
+ * being wired on this host (mirrors `lakeViewComplete`'s contract on the access resolver: a
11874
+ * failure must report unknown, never a false zero).
11875
+ *
11876
+ * COUNT AND REASON ONLY, DELIBERATELY. Never a lake id, name, or tag: the caller may not be
11877
+ * permitted to know a given excluded lake exists at all, and this field must stay safe to show
11878
+ * them regardless of which specific lake(s) it is counting. `reason` is a closed enum, not free
11879
+ * text - prose could leak a lake's identity through phrasing - so a future exclusion cause (e.g.
11880
+ * an archived or quota-limited lake) adds an enum value here rather than a description.
11881
+ *
11882
+ * 'access' is the only reason today: the caller's org membership or the lake's public listing
11883
+ * surfaced it as a candidate (they could see it exists) but they hold neither its own
11884
+ * gate/entitlement nor an ownership or grant exception for it.
11885
+ *
11886
+ * A session-preauthorized lake (unionPreauthorizedLakeAccess) that is ALSO gate-dropped from
11887
+ * this account-wide count is corrected, not merely narrow: the seed's targeted measurement
11888
+ * (measureIdentityNamedExclusion, ChatCompletionProcess's promptMeta seed) excludes exactly the
11889
+ * tags this turn successfully admitted via preauthorization before running the gate query, so an
11890
+ * admitted-and-searched lake never reports here as excluded. This account-wide number itself
11891
+ * (excludedByAccessCount on getDynamicDataLakeAccess) is still computed before that union and is
11892
+ * NOT corrected the same way - only the per-turn targeted measurement is, which is what a
11893
+ * preauthorized session's own narrowing always uses (see sessionNamesALake's call site).
11894
+ */
11895
+ excludedLakes: z$1.object({
11896
+ count: z$1.number().int().nonnegative(),
11897
+ reason: z$1.enum(["access"])
11898
+ }).optional()
12465
11899
  });
12466
11900
  /**
12467
11901
  * Why a grounded turn's library scan stopped short of the whole library.
@@ -12669,7 +12103,7 @@ const MeSubscriptionSchema = z$1.object({
12669
12103
  /** ISO 8601. When the current billing period ends - not a cancellation date. */
12670
12104
  current_period_ends_at: z$1.string()
12671
12105
  });
12672
- z$1.object({
12106
+ const MeResponseSchema = z$1.object({
12673
12107
  /** Stable B4M user id. Safe to key an integrator's own records on. */
12674
12108
  id: z$1.string(),
12675
12109
  /** Display name. Never the email address. */
@@ -13319,13 +12753,6 @@ const DATA_LAKES = [{
13319
12753
  }
13320
12754
  })()];
13321
12755
  new Set(DATA_LAKES.map((l) => l.id));
13322
- /**
13323
- * Canonical normalization for entitlement keys + `requiredEntitlement` values - the ONE
13324
- * rule, applied at write time (create/update/stamp) and at match time. Mirrors the
13325
- * entitlement registry's `normalizeTag` (trim + lowercase) so a value authored in any
13326
- * casing matches the lowercase keys the resolver produces.
13327
- */
13328
- const normalizeEntitlementKey = (key) => key.trim().toLowerCase();
13329
12756
  const sha256Regex = /^[a-f0-9]{64}$/;
13330
12757
  const requiredUserTagValue = z.string().trim().min(1).max(100).refine((s) => !/[,;]/.test(s), "User tag must be a single tag with no commas or semicolons (e.g. \"vip\" or \"Sales Team\")");
13331
12758
  z.object({
@@ -13335,7 +12762,8 @@ z.object({
13335
12762
  fileTagPrefix: z.string().trim().min(2).max(30).refine((s) => s.endsWith(":"), "Tag prefix must end with \":\" (e.g. \"acme:\")").refine((s) => !hasBlankTagPrefixSegment(s), "Tag prefix segments must be non-empty (e.g. \"acme:\" or \"acme:legal:\")").refine((s) => !isReservedTagPrefix(s), `Tag prefix cannot use the reserved "${DATALAKE_TAG_PREFIX}" namespace`),
13336
12763
  requiredUserTag: requiredUserTagValue.optional(),
13337
12764
  requiredEntitlement: z.string().min(3).max(100).refine((s) => s.includes(":") && s.split(":").every((part) => part.length > 0), "Entitlement key must be namespaced with non-empty parts (e.g. \"product:pro\")").optional(),
13338
- organizationId: z.string().optional()
12765
+ organizationId: z.string().optional(),
12766
+ origin: z.enum(DATA_LAKE_ORIGINS).optional()
13339
12767
  });
13340
12768
  z.object({
13341
12769
  name: z.string().min(1).max(200).optional(),
@@ -13347,7 +12775,8 @@ z.object({
13347
12775
  requiredEntitlement: z.union([z.literal(""), z.string().min(3).max(100).refine((s) => s.includes(":") && s.split(":").every((part) => part.length > 0), "Entitlement key must be namespaced with non-empty parts (e.g. \"product:pro\")")]).optional(),
13348
12776
  auditQueryTextEnabled: z.boolean().optional(),
13349
12777
  lakeMemoryEnabled: z.boolean().optional(),
13350
- requiredPassageTokenTarget: z.number().int().min(64).max(OVERSIZED_PASSAGE_TOKEN_THRESHOLD).nullable().optional()
12778
+ requiredPassageTokenTarget: z.number().int().min(64).max(OVERSIZED_PASSAGE_TOKEN_THRESHOLD).nullable().optional(),
12779
+ origin: z.enum(DATA_LAKE_ORIGINS).optional()
13351
12780
  });
13352
12781
  z.object({
13353
12782
  groundingMode: z.enum(DATA_LAKE_GROUNDING_MODES).optional(),
@@ -13489,7 +12918,380 @@ z$1.object({
13489
12918
  model: ImageTemplateModelSchema,
13490
12919
  settings: ImageTemplateSettingsSchema
13491
12920
  }).omit({ model: true }).partial();
13492
- z$1.union([ToolExecutionResponseSchema, ApiErrorSchema]);
12921
+ /**
12922
+ * Identity factory that pins an endpoint contract's type. The `const` type
12923
+ * parameter preserves the concrete request-schema type so downstream adapters
12924
+ * can infer the validated body type (`z.infer<contract['request']>`) rather than
12925
+ * collapsing to the `z.ZodTypeAny` constraint.
12926
+ *
12927
+ * Otherwise deliberately a no-op at runtime (no registry side effect, no
12928
+ * `.openapi()`) so a contract stays a plain, transport-agnostic value that any
12929
+ * runtime can import - the one exception is the `pathParams`/`queryParams`
12930
+ * overlap check below, a plain assertion with no side effect of its own.
12931
+ */
12932
+ function defineEndpoint(contract) {
12933
+ if (contract.pathParams && contract.queryParams) {
12934
+ const queryKeys = new Set(Object.keys(contract.queryParams.shape));
12935
+ const overlap = Object.keys(contract.pathParams.shape).filter((key) => queryKeys.has(key));
12936
+ if (overlap.length > 0) throw new Error(`defineEndpoint(${contract.operationId}): pathParams and queryParams both declare ${JSON.stringify(overlap)}. Next merges both into req.query by name, so the path segment would silently win over the query value. Rename one side.`);
12937
+ }
12938
+ return contract;
12939
+ }
12940
+ defineEndpoint({
12941
+ method: "post",
12942
+ path: "/api/chat",
12943
+ operationId: "sendChatMessage",
12944
+ summary: "Send a chat message",
12945
+ description: "Sends a message to the AI and creates a quest to process it. By default (async) the call returns immediately with a quest id; poll `GET /api/quests/{id}` for the reply. Send `wait: true` to block until the reply is ready and receive it inline. A tool that produced machine-readable state reports it under `toolPayloads` - an array of `{ type, payload }` entries in emission order, alongside (never instead of) the prose reply - on the `wait: true` body and on the polled quest. A turn can FAIL after the ACK - reported on the polled quest as `type: \"error\"`, since the ACK was already sent. A terminal `status: \"stopped\"` (a missing session, a user-cancelled turn) is ALSO a failure even without `type: \"error\"` - it carries an explanatory string in `reply`/`replies` rather than an answer. `type` is the failure signal for the classified failure classes (an abort, a provider timeout, credit exhaustion); `errorCode` is an optional refinement present only when the failure is a classified billing reason. A recovered stuck quest is NOT in that list: one that still has renderable content resolves as a success by design. `QUEST_ERROR_CODES` has two members, but only `insufficient_credits` is raised as a quest errorCode by any current throw site on this endpoint - `spend_cap_exceeded` is thrown only by the embed chat route's pre-flight, which fires outside the process try/catch that would classify it onto a quest. A caller must treat `type: \"error\"` OR a terminal `status: \"stopped\"` as failure even when `errorCode` is absent, and must not read `reply` as an answer without checking those first. Authenticate with an API key (`b4m_live_`) or a JWT.",
12946
+ tags: ["AI"],
12947
+ auth: "apiKeyOrJwt",
12948
+ scopes: ["ai:chat", "ai:generate"],
12949
+ request: SimplifiedChatRequestSchema,
12950
+ requestExample: {
12951
+ message: "How do I reset my password?",
12952
+ toolMode: "smart"
12953
+ },
12954
+ emitsRateLimitHeaders: true,
12955
+ responses: {
12956
+ 200: {
12957
+ description: "Message accepted - NOT a completed turn. The default (async) path returns this queued ACK; the outcome arrives on `GET /api/quests/{id}` (see the `sendChatMessage200PollResult` schema). With `wait: true` the body additionally carries the completed reply (`response`/`responses`), `toolPayloads`, `createdAt`, and `performance` timings - fields not modelled here yet; the synchronous response shape is a follow-up. A turn that FAILS still resolves with `200`, never a 4xx, on both that `wait: true` body and the polled quest (`GET /api/quests/{id}`) - the prose explaining why lands in `reply`/`response` like any other answer, so the reply text alone cannot tell a failure from an answer. `type` is the field that can: both surfaces carry it unconditionally, so match on `type: \"error\"` first - it covers credit exhaustion, a provider timeout or overload, and an in-process aborted turn; a real answer carries the turn's actual completion type instead (`\"message\"` for an ordinary reply). Two related states do NOT set `type: \"error\"`: a user-cancelled turn resolves as `status: \"stopped\"` with `type` left at `\"message\"`, and a recovered stuck quest that still has renderable content resolves as a success (`status: \"done\"`, no error) by design, to avoid destroying content to report a failure. `errorCode` then names the failure reason, but only for the billing failures that have one - `\"insufficient_credits\"` today; it is absent on every other `type: \"error\"` turn, so never use its absence to infer success. On a real answer `errorCode` is absent from the `wait: true` body. Contrast the tts/music/soundEffects contracts, which reject synchronously with a 422 carrying the same `errorCode` vocabulary.",
12958
+ schema: ChatAckSchema,
12959
+ pollResult: {
12960
+ schema: ChatQuestPollResultSchema,
12961
+ description: "Outcome fields of the quest polled at `GET /api/quests/{id}` after this ACK. A finished turn that failed is `status: \"done\"` with `type: \"error\"` and the failure text in `reply`, so a caller reading `reply` alone cannot tell a failure from an answer - check `type` first, and also treat a terminal `status: \"stopped\"` (a missing session, a user-cancelled turn) as failure even though it never sets `type`. `errorCode` is an optional refinement of `type: \"error\"`, present only for a classified billing failure; credit exhaustion arrives here as `insufficient_credits`, the same vocabulary the synchronous 422s on `/api/ai/music`, `/api/ai/sound-effects` and `/api/ai/tts` use. `QUEST_ERROR_CODES` publishes a second member, `spend_cap_exceeded`, but no current throw site on this endpoint raises it as a quest errorCode: its only one, the embed chat route's pre-flight 422, fires outside the process try/catch that would classify it onto the quest. Most `type: \"error\"` turns - an abort, a provider timeout or overload - have NO `errorCode`; its absence does not mean success, only that the failure is unclassified. A recovered stuck quest that still has renderable content is not a failure at all: it keeps `type: \"message\"` even though it did not finish, so a caller gets the content rather than an error. The poll body carries further fields not modelled here, including `images`, `files`, `toolPayloads`, `promptMeta`, and the attachment report (`attachmentNotices`/`attachmentDelivery`) - only the outcome subset is modelled here.",
12962
+ example: {
12963
+ id: "664f1c2b9a1e4d0012ab34cd",
12964
+ status: "done",
12965
+ type: "error",
12966
+ errorCode: "insufficient_credits",
12967
+ reply: "You're out of credits. This request needs about 12 credits, but only 3 are available."
12968
+ }
12969
+ }
12970
+ },
12971
+ 400: {
12972
+ description: "No usable default chat model is configured and none was supplied.",
12973
+ schema: ApiErrorSchema
12974
+ },
12975
+ 404: {
12976
+ description: "No notebook/session exists to attach the message to.",
12977
+ schema: ApiErrorSchema
12978
+ },
12979
+ 422: {
12980
+ description: "Request body failed schema validation.",
12981
+ schema: ApiErrorSchema
12982
+ },
12983
+ 429: {
12984
+ description: "Per-user rate limit exceeded.",
12985
+ schema: ApiErrorSchema
12986
+ }
12987
+ },
12988
+ codeSample: {
12989
+ authToken: "b4m_live_<key>",
12990
+ streaming: false,
12991
+ body: {
12992
+ message: "How do I reset my password?",
12993
+ toolMode: "smart"
12994
+ }
12995
+ }
12996
+ });
12997
+ defineEndpoint({
12998
+ method: "post",
12999
+ path: "/api/v1/agent-executions",
13000
+ operationId: "startAgentExecution",
13001
+ summary: "Start an agent execution",
13002
+ description: "Runs the tool-using agent (ReAct) loop against a session - the same pipeline the product UI's Agent Mode toggle dispatches to, and the REST equivalent of the `agent_execute` WebSocket command. The run is asynchronous: this returns `202` with an execution id, and the caller polls `GET /api/v1/agent-executions/{id}` until `status` is terminal (`completed`, `failed`, or `aborted`). Nothing is streamed back over REST - for live iteration events, use the WebSocket route instead. Omit `agent_id` to get the profile the session's own surface resolves to, which is what reproduces the in-app toggle. The final reply is also written to the session as a normal chat message, so it appears in history. Naming a tool in `tools` also PRE-APPROVES it for the run: there is no interactive client to answer a permission prompt, so a run that calls an approval-gated tool you did not name fails with that tool named in `error` rather than hanging. Authenticate with an API key (`b4m_live_`) or a JWT.",
13003
+ tags: ["AI"],
13004
+ auth: "apiKeyOrJwt",
13005
+ scopes: ["ai:chat", "ai:generate"],
13006
+ request: AgentExecutionStartRequestSchema,
13007
+ requestExample: {
13008
+ session_id: "<sessionId>",
13009
+ message: "Audit this data set and summarize what stands out."
13010
+ },
13011
+ emitsRateLimitHeaders: true,
13012
+ responses: {
13013
+ 202: {
13014
+ description: "Run accepted and dispatched. Poll `tracking_info.poll_url` for status and the answer.",
13015
+ schema: AgentExecutionAckSchema
13016
+ },
13017
+ 400: {
13018
+ description: "No usable default chat model is configured and none was supplied.",
13019
+ schema: ApiErrorSchema
13020
+ },
13021
+ 404: {
13022
+ description: "No session with the given `session_id` is owned by the caller, or `organization_id` names an organization the caller does not belong to. \"Exists but is not yours\" is reported as 404 too, so neither session ids nor org membership can be probed through this endpoint.",
13023
+ schema: ApiErrorSchema
13024
+ },
13025
+ 409: {
13026
+ description: "The caller already has the maximum number of agent runs in flight - a per-user cap shared with runs started from the product UI. Wait for one to reach a terminal status before starting another; aborting a run is currently only possible over the WebSocket route.",
13027
+ schema: ApiErrorSchema
13028
+ },
13029
+ 429: {
13030
+ description: "Per-user rate limit exceeded.",
13031
+ schema: ApiErrorSchema
13032
+ },
13033
+ 502: {
13034
+ description: "The executor could not be dispatched. The run did not start, so a retry is safe.",
13035
+ schema: ApiErrorSchema
13036
+ }
13037
+ },
13038
+ codeSample: {
13039
+ authToken: "b4m_live_<key>",
13040
+ streaming: false,
13041
+ body: {
13042
+ session_id: "<sessionId>",
13043
+ message: "Audit this data set and summarize what stands out."
13044
+ }
13045
+ }
13046
+ });
13047
+ defineEndpoint({
13048
+ method: "get",
13049
+ path: "/api/v1/agent-executions/{id}",
13050
+ operationId: "getAgentExecution",
13051
+ summary: "Get an agent execution",
13052
+ description: "Returns the status, reasoning trace, and (once terminal) the final answer of a run started by `POST /api/v1/agent-executions`. `steps` grows while the run is in flight, so polling this endpoint is also how a REST caller follows the loop. Safe (GET) requests on this route are exempt from the per-day API-key quota so polling a single run costs one daily slot, not one per poll; the per-minute burst limit still applies.",
13053
+ tags: ["AI"],
13054
+ auth: "apiKeyOrJwt",
13055
+ scopes: ["ai:chat", "ai:generate"],
13056
+ pathParams: AgentExecutionIdParamSchema,
13057
+ emitsRateLimitHeaders: true,
13058
+ responses: {
13059
+ 200: {
13060
+ description: "The execution, its trace, and its answer if it has finished.",
13061
+ schema: AgentExecutionStatusResponseSchema
13062
+ },
13063
+ 404: {
13064
+ description: "No execution with that id is visible to the caller.",
13065
+ schema: ApiErrorSchema
13066
+ },
13067
+ 429: {
13068
+ description: "Per-user rate limit exceeded.",
13069
+ schema: ApiErrorSchema
13070
+ }
13071
+ },
13072
+ codeSample: {
13073
+ authToken: "b4m_live_<key>",
13074
+ streaming: false,
13075
+ body: {}
13076
+ }
13077
+ });
13078
+ defineEndpoint({
13079
+ method: "put",
13080
+ path: "/api/sessions/{id}",
13081
+ operationId: "updateSession",
13082
+ summary: "Update a session",
13083
+ description: "Updates a session (called a \"notebook\" in the product UI): its name, attached knowledge files, tags, or retrieval settings. Set `knowledgeIds` and `forceKnowledgeRetrieval: true` together to enable grounded retrieval for `POST /api/chat` against this session - retrieval is gated by these session fields, not by the chat request. `lakeScope` narrows that retrieval to a chosen set of data lakes; omit it to leave the current choice unchanged, or send `null` to clear it so retrieval reaches every lake you can access. Authenticate with an API key (`b4m_live_`) or a JWT. Warning: adding to `knowledgeIds` shares those files with every member of every project containing this session by default (see `propagateToProjects`), and that sharing cannot be undone through the UI.",
13084
+ tags: ["Sessions"],
13085
+ auth: "apiKeyOrJwt",
13086
+ scopes: ["notebooks:write"],
13087
+ pathParams: SessionIdParamSchema,
13088
+ request: SessionUpdateRequestSchema,
13089
+ emitsRateLimitHeaders: true,
13090
+ requestExample: {
13091
+ knowledgeIds: ["<fabFileId>"],
13092
+ forceKnowledgeRetrieval: true
13093
+ },
13094
+ responses: {
13095
+ 200: {
13096
+ description: "The updated session.",
13097
+ schema: SessionResponseSchema
13098
+ },
13099
+ 404: {
13100
+ description: "No session exists with the given id.",
13101
+ schema: ApiErrorSchema
13102
+ }
13103
+ },
13104
+ codeSample: {
13105
+ authToken: "b4m_live_<key>",
13106
+ streaming: false,
13107
+ body: {
13108
+ knowledgeIds: ["<fabFileId>"],
13109
+ forceKnowledgeRetrieval: true
13110
+ }
13111
+ }
13112
+ });
13113
+ defineEndpoint({
13114
+ method: "post",
13115
+ path: "/api/ai/v1/tools",
13116
+ operationId: "executeTool",
13117
+ summary: "Execute a server-side tool",
13118
+ description: "Runs one of the built-in server-side tools (`weather_info`, `web_search`, `web_fetch`) and returns its result as JSON. Authenticate with a JWT access token only - API keys are NOT accepted on this endpoint. Rate-limited to 100 requests/hour. `request_id` echoes the X-Request-ID response header.",
13119
+ tags: ["AI"],
13120
+ auth: "jwtOnly",
13121
+ request: ToolExecutionRequestSchema,
13122
+ requestExample: {
13123
+ toolName: "web_search",
13124
+ input: { query: "how to reset a password" }
13125
+ },
13126
+ responses: {
13127
+ 200: {
13128
+ description: "Tool executed successfully (`success` is always true here).",
13129
+ schema: ToolExecutionResponseSchema,
13130
+ example: {
13131
+ success: true,
13132
+ result: { summary: "Top results for the query." },
13133
+ executionTimeMs: 842,
13134
+ request_id: "abc-123"
13135
+ }
13136
+ },
13137
+ 400: {
13138
+ description: "Malformed JSON body.",
13139
+ schema: ApiErrorSchema
13140
+ },
13141
+ 401: {
13142
+ description: "Missing or invalid JWT (an API key is rejected here).",
13143
+ schema: ApiErrorSchema
13144
+ },
13145
+ 429: {
13146
+ description: "Rate limit exceeded (100 requests/hour).",
13147
+ schema: ApiErrorSchema
13148
+ },
13149
+ 500: {
13150
+ description: "Tool execution failed (`success: false` with `error`) or an unexpected server error.",
13151
+ schema: z$1.union([ToolExecutionResponseSchema, ApiErrorSchema]),
13152
+ bespokeErrorShape: "A tool that ran but failed returns the full ToolExecutionResponse (success: false), not an error envelope."
13153
+ }
13154
+ },
13155
+ codeSample: {
13156
+ authToken: "<access_token>",
13157
+ streaming: false,
13158
+ body: {
13159
+ toolName: "web_search",
13160
+ input: { query: "how to reset a password" }
13161
+ }
13162
+ }
13163
+ });
13164
+ defineEndpoint({
13165
+ method: "post",
13166
+ path: "/api/ai/v1/completions",
13167
+ operationId: "createCompletion",
13168
+ summary: "Create a chat completion",
13169
+ description: "OpenAI-compatible completion. The response is ALWAYS an SSE stream (`text/event-stream`), regardless of the `stream` flag: a `meta` event, then `content`/`tool_use` events carrying `usage`/`credits`, terminated by `data: [DONE]`. Once the stream has opened the HTTP status stays 200 and failures arrive as an in-band `error` event. The terminal event carries `stopReason`; treat `max_tokens` as a TRUNCATED reply rather than a complete one, and note that omitting `max_tokens` on the request lets the server size the output ceiling for the model (recommended for reasoning models, which spend thinking tokens inside that ceiling). A message `content` may be a string or an array of parts; image parts are accepted in OpenAI Chat (`image_url`), OpenAI Responses (`input_image`) or Anthropic (`image` with a `source`) form and are translated to whatever the target model speaks. Authenticate with an API key (`b4m_live_`) or a JWT.\n\nBILLING FAILURES. Headers are flushed before authentication or pricing, so unlike the JSON surfaces this endpoint has no pre-stream `422` + `errorCode: \"insufficient_credits\"` to pair with: credit exhaustion ALWAYS arrives as the in-band `error` event, whether it is caught by the reservation before the first token or by settlement mid-generation. Branch on that event's `code` (`insufficient_credits` - buy credits; `spend_cap_exceeded` - the owner is solvent but this key hit its admin-set ceiling, so raise the cap), never on `message`, which is prose and may change. `code` is absent on unclassified failures.",
13170
+ tags: ["AI"],
13171
+ auth: "apiKeyOrJwt",
13172
+ scopes: ["ai:chat", "ai:generate"],
13173
+ request: CompletionRequestSchema,
13174
+ requestExample: {
13175
+ model: "claude-opus-4-8",
13176
+ messages: [{
13177
+ role: "user",
13178
+ content: "How do I reset my password?"
13179
+ }],
13180
+ max_tokens: 500
13181
+ },
13182
+ streaming: true,
13183
+ responses: {
13184
+ 200: {
13185
+ description: "SSE stream of completion events (see the CompletionStreamEvent shape).",
13186
+ contentType: "text/event-stream",
13187
+ schema: CompletionStreamEventSchema,
13188
+ example: {
13189
+ type: "content",
13190
+ text: "To reset your password, click 'Forgot password' on the login screen.",
13191
+ usage: {
13192
+ inputTokens: 42,
13193
+ outputTokens: 12
13194
+ },
13195
+ credits: {
13196
+ used: 1,
13197
+ usdCost: 37e-5
13198
+ },
13199
+ stopReason: "end_turn"
13200
+ }
13201
+ },
13202
+ 400: {
13203
+ description: "Malformed JSON body (rejected before the stream opens).",
13204
+ schema: ApiErrorSchema
13205
+ }
13206
+ },
13207
+ codeSample: {
13208
+ authToken: "b4m_live_<key>",
13209
+ streaming: true,
13210
+ body: {
13211
+ model: "claude-opus-4-8",
13212
+ messages: [{
13213
+ role: "user",
13214
+ content: "How do I reset my password?"
13215
+ }],
13216
+ max_tokens: 500
13217
+ }
13218
+ }
13219
+ });
13220
+ defineEndpoint({
13221
+ method: "post",
13222
+ path: "/api/ai/tts",
13223
+ operationId: "synthesizeSpeech",
13224
+ summary: "Synthesize speech from text",
13225
+ description: "Generates speech from text using OpenAI or ElevenLabs. The default `encoding: \"binary\"` streams raw audio bytes with an `audio/*` Content-Type; `encoding: \"base64\"` returns JSON instead. When the requested provider has no usable key (or the provider rejects it), another configured provider stands in and the substitution is reported via `provider`/`fallbackFrom` and the `X-B4M-Tts-Provider*` headers. Input length is capped per provider (OpenAI 4096 characters, ElevenLabs 10000), and an output `format` the chosen provider cannot produce is rejected with a 422 before any provider cost is incurred. Generated audio is saved to the file browser by default (opt out per-user via the saveGeneratedAudio preference, or per-call with `preview`); the outcome is reported via `saved`/`fabFileId` and the `X-B4M-Audio-*` headers. Authenticate with an API key (`b4m_live_`) or a JWT.",
13226
+ tags: ["Audio"],
13227
+ auth: "apiKeyOrJwt",
13228
+ scopes: ["ai:generate"],
13229
+ request: ttsRequestSchema,
13230
+ requestExample: {
13231
+ text: "Your password has been reset.",
13232
+ provider: "openai",
13233
+ voice: "alloy",
13234
+ format: "mp3"
13235
+ },
13236
+ responses: {
13237
+ 200: {
13238
+ description: "Speech synthesized. The default `encoding: \"binary\"` returns raw audio bytes whose Content-Type follows the requested `format` (`audio/mpeg` for mp3, else `audio/wav`, `audio/opus`, `audio/aac`, `audio/flac`, `audio/pcm`); `encoding: \"base64\"` returns the JSON body.",
13239
+ schema: ttsBase64ResponseSchema,
13240
+ example: {
13241
+ audio: "SUQzBAAAAAAA...",
13242
+ format: "mp3",
13243
+ contentType: "audio/mpeg",
13244
+ saved: true,
13245
+ fabFileId: "664f1c2b9a1e4d0012ab34cd"
13246
+ },
13247
+ alsoReturns: [
13248
+ { contentType: "audio/mpeg" },
13249
+ { contentType: "audio/wav" },
13250
+ { contentType: "audio/opus" },
13251
+ { contentType: "audio/aac" },
13252
+ { contentType: "audio/flac" },
13253
+ { contentType: "audio/pcm" }
13254
+ ],
13255
+ headers: {
13256
+ "X-B4M-Tts-Provider": "The provider that produced the audio. Present only when a fallback happened.",
13257
+ "X-B4M-Tts-Provider-Fallback-From": "The originally requested provider that could not serve the request. Present only on a fallback.",
13258
+ "X-B4M-Audio-Saved": "Whether a browsable copy was saved to the file browser (\"true\"/\"false\").",
13259
+ "X-B4M-Audio-Fab-File-Id": "Id of the saved file. Present only when the copy was saved."
13260
+ }
13261
+ },
13262
+ 401: {
13263
+ description: "Missing/invalid credentials, or no provider has a usable key (`provider_not_configured`).",
13264
+ schema: ttsErrorResponseSchema
13265
+ },
13266
+ 413: {
13267
+ description: "The audio was generated (and billed) but is too large to return over this endpoint. Retrieve it from `fileUrl` when a browsable copy was saved.",
13268
+ schema: ttsResponseTooLargeSchema
13269
+ },
13270
+ 422: {
13271
+ description: "Request body failed validation, the text exceeds the provider character limit, the provider cannot produce the requested `format`, or the caller cannot afford the synthesis - the last of those is the only one tagged `errorCode: \"insufficient_credits\"`, so match on the classifier rather than the status to tell a billing failure from a bad request.",
13272
+ schema: ttsErrorResponseSchema
13273
+ },
13274
+ 429: {
13275
+ description: "The provider rate-limited the request.",
13276
+ schema: ttsErrorResponseSchema
13277
+ },
13278
+ 502: {
13279
+ description: "The provider failed to generate speech.",
13280
+ schema: ttsErrorResponseSchema
13281
+ }
13282
+ },
13283
+ emitsRateLimitHeaders: true,
13284
+ codeSample: {
13285
+ authToken: "b4m_live_<key>",
13286
+ streaming: false,
13287
+ body: {
13288
+ text: "Your password has been reset.",
13289
+ provider: "openai",
13290
+ voice: "alloy",
13291
+ encoding: "base64"
13292
+ }
13293
+ }
13294
+ });
13493
13295
  /**
13494
13296
  * Response details shared by the endpoints that return generated audio as raw
13495
13297
  * bytes (music, sound effects). Not a contract - just the pieces both of their
@@ -13526,8 +13328,179 @@ const generatedAudioBody = () => ({
13526
13328
  alsoReturns: GENERATED_AUDIO_CONTENT_TYPES.slice(1).map((contentType) => ({ contentType })),
13527
13329
  headers: GENERATED_AUDIO_SAVE_HEADERS
13528
13330
  });
13529
- ({ ...generatedAudioBody() });
13530
- ({ ...generatedAudioBody() });
13331
+ defineEndpoint({
13332
+ method: "post",
13333
+ path: "/api/ai/music",
13334
+ operationId: "generateMusic",
13335
+ summary: "Generate background music",
13336
+ description: "Generates an instrumental or vocal background-music track from a text prompt and returns the raw audio bytes. `lengthMs` (3000-120000, default 10000) is forced on the provider, so the generated track always matches the billed length; credits are reserved before generation and refunded if it fails. Generated audio is saved to the file browser by default (opt out via the saveGeneratedAudio preference); the outcome is reported via the `X-B4M-Audio-Saved` / `X-B4M-Audio-Fab-File-Id` / `X-B4M-Audio-File-Url` response headers - use `X-B4M-Audio-File-Url` to fetch the saved copy, since `GET /api/files/{id}` fails closed until moderation completes. Authenticate with an API key (`b4m_live_`) or a JWT.",
13337
+ tags: ["Audio"],
13338
+ auth: "apiKeyOrJwt",
13339
+ scopes: ["ai:generate"],
13340
+ request: musicRequestSchema,
13341
+ requestExample: {
13342
+ prompt: "calm lo-fi study beat with soft piano",
13343
+ lengthMs: 3e4,
13344
+ forceInstrumental: true
13345
+ },
13346
+ responses: {
13347
+ 200: {
13348
+ description: "Raw audio bytes; the Content-Type follows the requested `format` (mp3 by default).",
13349
+ ...generatedAudioBody()
13350
+ },
13351
+ 400: {
13352
+ description: "The billing user or organization could not be resolved.",
13353
+ schema: ApiErrorSchema
13354
+ },
13355
+ 422: {
13356
+ description: "Request body failed validation, or the caller cannot afford the track - the latter is tagged `errorCode: \"insufficient_credits\"` (the balance is short, or the org member credit cap is exhausted).",
13357
+ schema: InsufficientCreditsErrorSchema
13358
+ },
13359
+ 502: {
13360
+ description: "The provider failed to generate the track; reserved credits are refunded.",
13361
+ schema: ApiErrorSchema
13362
+ },
13363
+ 503: {
13364
+ description: "No provider API key is configured for this deployment.",
13365
+ schema: ApiErrorSchema
13366
+ }
13367
+ },
13368
+ emitsRateLimitHeaders: true,
13369
+ codeSample: {
13370
+ authToken: "b4m_live_<key>",
13371
+ streaming: false,
13372
+ body: {
13373
+ prompt: "calm lo-fi study beat with soft piano",
13374
+ lengthMs: 3e4,
13375
+ forceInstrumental: true
13376
+ }
13377
+ }
13378
+ });
13379
+ defineEndpoint({
13380
+ method: "post",
13381
+ path: "/api/ai/sound-effects",
13382
+ operationId: "generateSoundEffect",
13383
+ summary: "Generate a sound effect",
13384
+ description: "Generates a short sound effect from a text description and returns the raw audio bytes. Omitting `durationSeconds` lets the provider pick the length (and bills at its default); `promptInfluence` trades prompt fidelity (1) against variation (0). Credits are reserved before generation and refunded if it fails. Generated audio is saved to the file browser by default (opt out via the saveGeneratedAudio preference); the outcome is reported via the `X-B4M-Audio-Saved` / `X-B4M-Audio-Fab-File-Id` / `X-B4M-Audio-File-Url` response headers - use `X-B4M-Audio-File-Url` to fetch the saved copy, since `GET /api/files/{id}` fails closed until moderation completes. Authenticate with an API key (`b4m_live_`) or a JWT.",
13385
+ tags: ["Audio"],
13386
+ auth: "apiKeyOrJwt",
13387
+ scopes: ["ai:generate"],
13388
+ request: soundEffectsRequestSchema,
13389
+ requestExample: {
13390
+ text: "heavy wooden door creaking open",
13391
+ durationSeconds: 3,
13392
+ promptInfluence: .5
13393
+ },
13394
+ responses: {
13395
+ 200: {
13396
+ description: "Raw audio bytes; the Content-Type follows the requested `format` (mp3 by default).",
13397
+ ...generatedAudioBody()
13398
+ },
13399
+ 400: {
13400
+ description: "The billing user or organization could not be resolved.",
13401
+ schema: ApiErrorSchema
13402
+ },
13403
+ 422: {
13404
+ description: "Request body failed validation, or the caller cannot afford the effect - the latter is tagged `errorCode: \"insufficient_credits\"` (the balance is short, or the org member credit cap is exhausted).",
13405
+ schema: InsufficientCreditsErrorSchema
13406
+ },
13407
+ 502: {
13408
+ description: "The provider failed to generate the effect; reserved credits are refunded.",
13409
+ schema: ApiErrorSchema
13410
+ },
13411
+ 503: {
13412
+ description: "No provider API key is configured for this deployment.",
13413
+ schema: ApiErrorSchema
13414
+ }
13415
+ },
13416
+ emitsRateLimitHeaders: true,
13417
+ codeSample: {
13418
+ authToken: "b4m_live_<key>",
13419
+ streaming: false,
13420
+ body: {
13421
+ text: "heavy wooden door creaking open",
13422
+ durationSeconds: 3,
13423
+ promptInfluence: .5
13424
+ }
13425
+ }
13426
+ });
13427
+ defineEndpoint({
13428
+ method: "get",
13429
+ path: "/api/v1/me",
13430
+ operationId: "getMe",
13431
+ summary: "Get the authenticated caller",
13432
+ description: "Returns the authenticated caller: stable id, display name, plan tier, personal credit balance, and entitlement keys. The subject is always the credential holder - this endpoint accepts no user id, owner id, or impersonation parameter of any kind, so a key can only ever read its own owner. `credits.balance` is the caller's personal ledger; a call billed to an organization draws on a pool this number does not describe. Gate on `tier != \"free\"` for \"is this caller paying\" and on `subscription.price_id` for which product - the `basic`/`pro` rungs come from an internal plan ladder and do not track a plan's marketing name. Responses are never cacheable. Authenticate with an API key (`b4m_live_`) carrying `me:read`, or a JWT.",
13433
+ tags: ["Account"],
13434
+ auth: "apiKeyOrJwt",
13435
+ scopes: ["me:read"],
13436
+ emitsRateLimitHeaders: true,
13437
+ responses: {
13438
+ 200: {
13439
+ description: "The caller, their tier, their personal credit balance, and their entitlement keys.",
13440
+ schema: MeResponseSchema,
13441
+ example: {
13442
+ id: "507f1f77bcf86cd799439011",
13443
+ name: "Ada Lovelace",
13444
+ tier: "basic",
13445
+ subscription: {
13446
+ plan_name: "Professional",
13447
+ price_id: "price_123",
13448
+ interval: "monthly",
13449
+ current_period_ends_at: "2026-10-18T00:00:00.000Z"
13450
+ },
13451
+ credits: { balance: 31667 },
13452
+ entitlements: ["base"]
13453
+ }
13454
+ },
13455
+ 429: {
13456
+ description: "Per-user rate limit exceeded.",
13457
+ schema: ApiErrorSchema
13458
+ }
13459
+ },
13460
+ codeSample: {
13461
+ authToken: "b4m_live_<key>",
13462
+ streaming: false,
13463
+ body: {}
13464
+ }
13465
+ });
13466
+ /**
13467
+ * The fence language the model writes to place an inline map of search-result places in a reply.
13468
+ *
13469
+ * Same contract as SEARCH_RESULT_CARDS_LANGUAGE (searchResultCards.ts): `\w`-only and lowercase,
13470
+ * taught by WEB_SEARCH_MAP_PROMPT (@bike4mind/services), rendered by the reply renderer
13471
+ * (apps/client .../Session/PromptReplies.tsx), and rewritten to a plain list for every other
13472
+ * surface by `stripSearchResultCardFences`.
13473
+ */
13474
+ const LOCATION_MAP_LANGUAGE = "b4m_map";
13475
+ function googleMapsSearchUrl(name, placeId) {
13476
+ const params = new URLSearchParams({
13477
+ api: "1",
13478
+ query: name
13479
+ });
13480
+ if (placeId?.startsWith("ChIJ")) params.set("query_place_id", placeId);
13481
+ return `https://www.google.com/maps/search/?${params.toString()}`;
13482
+ }
13483
+ /**
13484
+ * The fence language the model writes to place image cards inline in a reply.
13485
+ *
13486
+ * Two families of consumer must agree on this constant:
13487
+ * - WEB_SEARCH_CARDS_PROMPT (@bike4mind/services) teaches the model to emit it, and the reply
13488
+ * renderer (apps/client .../Session/PromptReplies.tsx) intercepts it to render cards instead
13489
+ * of raw text.
13490
+ * - Every other surface that reads reply markdown for an export, download, copy, publish, or
13491
+ * Slack delivery must strip the fenced block entirely with `stripSearchResultCardFences`
13492
+ * below, rather than passing the model-authored card JSON through verbatim. See that
13493
+ * function's own doc comment for the current list of call sites.
13494
+ *
13495
+ * MUST contain only `\w` characters: both the renderer (`/language-(\w+)/`) and the notebook
13496
+ * curation extractor (```` /```(\w+)?/ ````) capture the language with `\w+`, so a hyphen would
13497
+ * silently truncate this and the cards would never render or be skipped.
13498
+ *
13499
+ * MUST already be lowercase: some consumers lowercase the language they capture before
13500
+ * comparing, so an uppercase-containing value would pass the `\w`-only rule above while silently
13501
+ * breaking that comparison.
13502
+ */
13503
+ const SEARCH_RESULT_CARDS_LANGUAGE = "b4m_cards";
13531
13504
  z$1.enum(["user", "convergence"]).optional().catch(void 0), z$1.string().optional();
13532
13505
  /**
13533
13506
  * Blessed, self-hosted artifact library script paths (root-relative).
@@ -13589,50 +13562,6 @@ const OPTIONAL_DEP_BLESSED_SCRIPT_PATHS = Object.values({
13589
13562
  }
13590
13563
  }).map((d) => d.path);
13591
13564
  [...REACT_BLESSED_SCRIPT_PATHS, ...OPTIONAL_DEP_BLESSED_SCRIPT_PATHS];
13592
- function isGPTImageModel(model) {
13593
- if (!model) return false;
13594
- return OPENAI_IMAGE_MODELS.includes(model) || model.startsWith("gpt-image-");
13595
- }
13596
- /**
13597
- * Flux Ultra is the only BFL generation model driven by `aspect_ratio`; the Pro family takes
13598
- * discrete `width`/`height` and ignores aspect_ratio outright. Must stay in sync with the branch
13599
- * in `BFLImageService.generate`, which is what actually builds the request body.
13600
- */
13601
- function isBflUltraImageModel(model) {
13602
- return model === "flux-pro-1.1-ultra";
13603
- }
13604
- /** Returns true specifically for gpt-image-2 (including versioned snapshots like gpt-image-2-2026-04-21). */
13605
- function isGPTImage2Model(model) {
13606
- if (!model) return false;
13607
- return model === "gpt-image-2" || model.startsWith("gpt-image-2");
13608
- }
13609
- /**
13610
- * Whether a user can access a model. Access is any-of (mirrors the Q3b data-lake
13611
- * rule, `getAccessibleDataLakes`): a non-admin reaches the model via
13612
- * `allowedUserTags ∩ userTags` OR `allowedEntitlements ∩ entitlementKeys`.
13613
- *
13614
- * `entitlementKeys` is optional - when omitted/empty the entitlement branch is
13615
- * inert, so a model with no `allowedEntitlements` behaves exactly as before
13616
- * (tag-only). This lets a tag-less subscriber reach an entitlement-gated model
13617
- * while leaving every existing tag-gated model unchanged (zero regression).
13618
- *
13619
- * Pure + zero-dependency (only types + the shared key normalizer), so it lives
13620
- * in `@bike4mind/common` as the SINGLE source of truth - imported by the core
13621
- * `@bike4mind/utils` re-export (server/services) AND the client
13622
- * `useAccessibleModels` hook, which previously kept a hand-rolled twin "to avoid
13623
- * AWS SDK imports". Common is browser-safe, so there is no longer any reason to
13624
- * duplicate the logic.
13625
- */
13626
- function isModelAccessible(model, userTags, isAdmin = false, entitlementKeys = []) {
13627
- if (!model.enabled) return false;
13628
- if (isAdmin) return true;
13629
- const normalizedUserTags = userTags.map((tag) => tag.toLowerCase());
13630
- const normalizedAllowedTags = (model.allowedUserTags ?? []).map((tag) => tag.toLowerCase());
13631
- if (normalizedUserTags.some((tag) => normalizedAllowedTags.includes(tag))) return true;
13632
- const normalizedKeys = entitlementKeys.map(normalizeEntitlementKey);
13633
- const normalizedAllowedEntitlements = (model.allowedEntitlements ?? []).map(normalizeEntitlementKey);
13634
- return normalizedKeys.some((key) => normalizedAllowedEntitlements.includes(key));
13635
- }
13636
13565
  [...OPENAI_IMAGE_MODELS, ...GEMINI_IMAGE_MODELS];
13637
13566
  const DashboardParamsSchema = z$1.object({
13638
13567
  dashboardDataSources: z$1.array(z$1.object({
@@ -13759,13 +13688,6 @@ OpenAIImageGenerationInput.extend({
13759
13688
  image: z$1.string(),
13760
13689
  output_format: ImageOutputFormatSchema.nullable().optional()
13761
13690
  });
13762
- const isUnlimitedHistory = (historyCount) => historyCount === -1;
13763
- /**
13764
- * A usable number for callers that need one (page size, overflow math, telemetry). Unlimited
13765
- * history has no count, so it resolves to the default page size, which is what an unwindowed
13766
- * request has always actually fetched.
13767
- */
13768
- const resolveHistoryFetchLimit = (historyCount) => isUnlimitedHistory(historyCount) || historyCount == null ? 14 : historyCount;
13769
13691
  z$1.object({
13770
13692
  /** Notebook session ID */
13771
13693
  sessionId: z$1.string(),
@@ -14287,6 +14209,18 @@ const ArtifactVersionMetaSchema = z$1.object({
14287
14209
  }),
14288
14210
  sha256Index: z$1.string()
14289
14211
  });
14212
+ /** One no-sign-in share link. `token` is the capability itself and is stripped from every
14213
+ * serialized response, so it is optional here: an owner-facing read carries the metadata
14214
+ * (id to revoke by, timestamps, per-link view count) with no token at all. `id` is the
14215
+ * subdocument `_id`, rendered as a string. */
14216
+ const ShareTokenEntrySchema = z$1.object({
14217
+ id: z$1.string().optional(),
14218
+ token: z$1.string().optional(),
14219
+ createdAt: z$1.date().optional(),
14220
+ revokedAt: z$1.date().nullish(),
14221
+ viewCount: z$1.int().nonnegative().prefault(0),
14222
+ lastViewedAt: z$1.date().nullish()
14223
+ });
14290
14224
  /** Reserved slugs - must include every tier URL token so a slug can't shadow routing. */
14291
14225
  const RESERVED_SLUGS = [
14292
14226
  "api",
@@ -14410,6 +14344,10 @@ z$1.object({
14410
14344
  shareToken: z$1.string().optional(),
14411
14345
  /** When `shareToken` was last minted/rotated; drives the owner-facing "link created" surface. */
14412
14346
  shareTokenUpdatedAt: z$1.date().nullish(),
14347
+ /** Every share link ever minted, revoked ones included. Mirrors the two fields above during
14348
+ * the rollout and becomes the source of truth once the backfill has run everywhere; `token`
14349
+ * is stripped from serialized responses, so an owner-facing read sees only the metadata. */
14350
+ shareTokens: z$1.array(ShareTokenEntrySchema).prefault([]),
14413
14351
  /** Collaboration gate: who (among viewers) may annotate. Orthogonal to
14414
14352
  * `visibility`. Defaults to `none` (read-only) until the owner opts in. */
14415
14353
  commentPolicy: CommentPolicySchema.prefault("none"),
@@ -14440,6 +14378,9 @@ z$1.object({
14440
14378
  declaredApiEndpoints: z$1.array(z$1.string()).prefault([]),
14441
14379
  /** Rendered body snapshot for reply/fabfile viewer pages (markdown or text). */
14442
14380
  renderedBody: z$1.string().optional(),
14381
+ /** Snapshot of the source reply's citables (reply source only), so a `b4m_map` fence in
14382
+ * `renderedBody` can still resolve its place ids after the source Quest is edited or deleted. */
14383
+ citables: z$1.array(CitableSourceSchema).optional(),
14443
14384
  publishedAt: z$1.date(),
14444
14385
  previousVersionMeta: ArtifactVersionMetaSchema.optional(),
14445
14386
  /** Full version history (oldest to newest); each entry's bytes are archived at
@@ -14693,170 +14634,70 @@ z$1.object({
14693
14634
  createdAt: z$1.union([z$1.string(), z$1.date()]).optional(),
14694
14635
  updatedAt: z$1.union([z$1.string(), z$1.date()]).optional()
14695
14636
  });
14637
+ ClaudeArtifactMimeTypes.RECHARTS, ClaudeArtifactMimeTypes.MERMAID, ClaudeArtifactMimeTypes.LATTICE, ClaudeArtifactMimeTypes.BLOG_DRAFT, ClaudeArtifactMimeTypes.CHESS;
14638
+ [...IMAGE_SIZE_CONSTRAINTS.DALL_E_2.sizes, ...IMAGE_SIZE_CONSTRAINTS.DALL_E_3.sizes];
14696
14639
  /**
14697
- * Length of `text` in Unicode CODE POINTS - the unit `IFabFileChunk.charLength` is stored in.
14698
- * Matches MongoDB's `$strLenCP`, which is what lets the char-length backfill
14699
- * (packages/scripts/datalake/backfill-chunk-char-length.ts) compute the same number server-side
14700
- * without reading chunk text out of the database. Deliberately NOT `text.length` (UTF-16 code
14701
- * units): the two differ on astral characters (surrogate pairs), and the write path and the
14702
- * backfill must agree exactly.
14703
- */
14704
- const countCodePoints = (text) => {
14705
- let count = 0;
14706
- for (const _ch of text) count++;
14707
- return count;
14708
- };
14709
- /**
14710
- * Default character budget for a super-linear parse pass. A starting point only -
14711
- * see the sizing rule below; any call site in front of a measured parser should pass
14712
- * an explicit `max` derived from that parser's own curve.
14640
+ * Zero-width space spliced into a defanged marker. It is invisible wherever the text is
14641
+ * rendered, so escaping never changes what the reader sees - only what the parser matches.
14713
14642
  */
14714
- const DEFAULT_PARSE_CAP = 1e5;
14643
+ const ZERO_WIDTH_SPACE = "​";
14715
14644
  /**
14716
- * Bound the length of input a super-linear parser will scan.
14645
+ * Defangs think-marker-shaped substrings inside provider-authored reasoning text so they can
14646
+ * never be mistaken for the real control markers adapters wrap around that same text.
14717
14647
  *
14718
- * A regex that backtracks polynomially (or worse) turns an unbounded input length
14719
- * into unbounded CPU on a shared process - the algorithmic-DoS shape. Refusing to
14720
- * scan more than `max` characters converts that into a fixed worst case regardless
14721
- * of how pathological the input is.
14722
- *
14723
- * Truncation (never throwing) is deliberate: these caps sit in front of best-effort
14724
- * cleaners and content detectors where an over-cap input should degrade quietly,
14725
- * not crash the pipeline.
14726
- *
14727
- * PREFER LINEARIZING THE PARSER. A cap is the fallback, not the first move. The email
14728
- * cleaners in services/lib/turndown.ts and the snippet-meta extractor in common/utils.ts
14729
- * both started here and both ended up as single-pass scans instead, because a cap that
14730
- * was small enough to bound their cost was also small enough to stop them working on
14731
- * real content. When the parser is linear, the cap can go.
14732
- *
14733
- * SIZING RULE - if you do cap, pick `max` from the parser's measured cost AT the cap,
14734
- * not from the headroom above real input. For an O(n^2)/O(n^3) inner loop, "far above
14735
- * anything legitimate" is not a bound: the email cleaners were cubic, so a 512k cap
14736
- * still ran for minutes. Measure the worst-case shape at the cap you intend to ship and
14737
- * make sure the number of milliseconds is one you can defend.
14738
- *
14739
- * Four further traps this helper cannot solve for you:
14740
- * - A per-item cap is not a total bound when the item count is also attacker-chosen
14741
- * (a .pptx picks its own slide count). Carry a running budget across the loop.
14742
- * - Cap what is SCANNED, not what is returned. Where the result is rendered or
14743
- * stored, re-append `input.slice(max)` unchanged: truncating the value silently
14744
- * drops content, and a cut inside an element can strand its closing tag. Splicing
14745
- * the original halves back together also avoids leaving a lone surrogate, since
14746
- * `slice` cuts by UTF-16 code unit.
14747
- * - Splicing the tail back is not enough when the cap changes the SHAPE of the parse
14748
- * rather than only its length - a pattern terminated by `$`, say, whose sections then
14749
- * end early and get classified differently. Check what an over-cap input parses INTO,
14750
- * not just that its characters survive.
14751
- * - Cap in front of a best-effort cleaner or detector, where degrading quietly is
14752
- * acceptable. Do not cap in front of a parser whose output drives execution (a
14753
- * tool-call parser): silently parsing a prefix there means running a subset of what
14754
- * was asked. If such a parser caps internally, its callers must pass it text already
14755
- * scoped to the construct being parsed.
14756
- *
14757
- * @returns the input unchanged when within budget, otherwise its first `max` chars.
14758
- */
14759
- function capForParse(input, max = DEFAULT_PARSE_CAP) {
14760
- if (!Number.isFinite(max) || max < 0) throw new RangeError(`capForParse: max must be a non-negative finite number, got ${max}`);
14761
- return input.length <= max ? input : input.slice(0, max);
14762
- }
14763
- /**
14764
- * Exactly 'WIDTHxHEIGHT' - nothing else, no surrounding whitespace and no third segment.
14765
- * Anchored on purpose: this module is the single source of the size rule, so a value that
14766
- * only looks like a resolution ('1024x1024x1024') must not be measured as one.
14767
- */
14768
- const SIZE_PATTERN = /^(\d+)x(\d+)$/;
14769
- /**
14770
- * Splits a 'WIDTHxHEIGHT' string into numeric edges, or null when it is not a pair
14771
- * of positive numbers. Lets callers tell a resolution that breaks a rule apart from
14772
- * a value that expresses no resolution at all ('auto', 'wide', undefined).
14773
- */
14774
- function parseSizeEdges(size) {
14775
- if (typeof size !== "string") return null;
14776
- const match = SIZE_PATTERN.exec(size);
14777
- if (!match) return null;
14778
- const width = Number(match[1]);
14779
- const height = Number(match[2]);
14780
- if (!width || !height) return null;
14781
- return {
14782
- width,
14783
- height
14784
- };
14785
- }
14786
- /**
14787
- * True when a custom gpt-image-2 resolution meets OpenAI's documented limits.
14788
- * gpt-image-2 accepts any resolution satisfying these, not only the presets in
14789
- * IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.sizes, so a flat preset check would reject
14790
- * valid custom sizes.
14791
- */
14792
- function satisfiesGptImage2Constraints({ width, height }) {
14793
- const { maxEdge, minTotalPixels, maxTotalPixels, edgeMultiple, maxAspectRatio } = IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.constraints;
14794
- const longEdge = Math.max(width, height);
14795
- const shortEdge = Math.min(width, height);
14796
- const totalPixels = width * height;
14797
- return longEdge <= maxEdge && width % edgeMultiple === 0 && height % edgeMultiple === 0 && longEdge / shortEdge <= maxAspectRatio && totalPixels >= minTotalPixels && totalPixels <= maxTotalPixels;
14648
+ * A reasoning delta is provider output, not our control plane - a model can say `<think>` or
14649
+ * `</think>` as literal content (reasoning about the protocol itself, or a leading/trailing
14650
+ * `</think>` from a provider that already delimits its own monologue). Adapters wrap the whole
14651
+ * delta in real markers via plain string concatenation, so an unescaped literal is
14652
+ * indistinguishable from a genuine open/close once it lands in the same string. Call this on
14653
+ * every raw reasoning delta before it is concatenated with THINK_OPEN_TAG/THINK_CLOSE_TAG.
14654
+ */
14655
+ function escapeThinkMarkers(text) {
14656
+ if (!text) return text;
14657
+ return text.replace(/<(\/?)think>/g, `<${ZERO_WIDTH_SPACE}$1think>`);
14798
14658
  }
14799
- /**
14800
- * Sizes the legacy (pre-GPT-Image) OpenAI generate endpoint accepts. The two tiers
14801
- * overlap on 1024x1024; the repeated entry is harmless because this list is only ever
14802
- * tested for membership.
14803
- */
14804
- const OPENAI_LEGACY_IMAGE_SIZES = [...IMAGE_SIZE_CONSTRAINTS.DALL_E_2.sizes, ...IMAGE_SIZE_CONSTRAINTS.DALL_E_3.sizes];
14805
- /**
14806
- * True when `size` may be sent to OpenAI for `model`. This is the single source of the
14807
- * OpenAI image size rule - generate, edit, the variation endpoint, the cost estimate and
14808
- * the settings UI all resolve through it, so a tier changing its accepted sizes is a
14809
- * one-line edit to IMAGE_SIZE_CONSTRAINTS rather than a hunt through retyped lists.
14810
- *
14811
- * gpt-image-2 takes 'auto', its presets, or any custom WIDTHxHEIGHT meeting
14812
- * satisfiesGptImage2Constraints. The gpt-image-1 family is limited to its three fixed
14813
- * sizes. Anything else is treated as legacy dall-e.
14814
- *
14815
- * OpenAI image models only: BFL, Gemini and xAI sizes are validated by their own adapters,
14816
- * and passing one of those models here would measure it against the wrong list.
14817
- */
14818
- function isSupportedImageSize(model, size) {
14819
- if (typeof size !== "string") return false;
14820
- if (isGPTImage2Model(model)) {
14821
- if (size === IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.autoSize || IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.sizes.includes(size)) return true;
14822
- const edges = parseSizeEdges(size);
14823
- return edges !== null && satisfiesGptImage2Constraints(edges);
14659
+ /** One less than the longer marker's length: the most characters a real marker prefix can span. */
14660
+ const MAX_PARTIAL_MARKER_LENGTH = 7;
14661
+ /** Length of the longest suffix of `text` that is a proper prefix of either marker token. */
14662
+ function partialMarkerSuffixLength(text) {
14663
+ const max = Math.min(MAX_PARTIAL_MARKER_LENGTH, text.length);
14664
+ for (let len = max; len > 0; len--) {
14665
+ const suffix = text.slice(-len);
14666
+ if ("<think>".startsWith(suffix) || "</think>".startsWith(suffix)) return len;
14824
14667
  }
14825
- if (isGPTImageModel(model)) return IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_1.sizes.includes(size);
14826
- return OPENAI_LEGACY_IMAGE_SIZES.includes(size);
14668
+ return 0;
14827
14669
  }
14828
14670
  /**
14829
- * The size to fall back to when a requested size is unsupported for `model`. Every tier
14830
- * defaults to 1024x1024 today; the indirection keeps that a per-tier decision.
14831
- */
14832
- function fallbackImageSize(model) {
14833
- if (isGPTImage2Model(model)) return IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.defaultSize;
14834
- if (isGPTImageModel(model)) return IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_1.defaultSize;
14835
- return IMAGE_SIZE_CONSTRAINTS.DALL_E_2.defaultSize;
14836
- }
14837
- /**
14838
- * The size the generate endpoint should send for a GPT-Image `model`, given what the
14839
- * caller asked for. Returns the input unchanged when nothing needs correcting, so a
14840
- * caller can compare the two to decide whether to warn.
14841
- *
14842
- * Absent: gpt-image-2 gets 'auto' - it is the only tier that can pick its own - and the
14843
- * gpt-image-1 family gets its fixed default.
14671
+ * Stateful counterpart to escapeThinkMarkers for text that arrives in streamed pieces.
14844
14672
  *
14845
- * Present but unsupported: replaced by the tier fallback, with one deliberate exception.
14846
- * A gpt-image-2 value that names no resolution at all ('wide') is forwarded untouched:
14847
- * gpt-image-2 sizing is open-ended, so an unrecognized token is left for OpenAI to
14848
- * interpret rather than second-guessed here. The gpt-image-1 family has fixed sizes and
14849
- * no such pass-through.
14673
+ * escapeThinkMarkers alone is only safe on a complete string: adapters call it once per
14674
+ * delta, but a provider is free to split a marker-shaped substring across two adjacent
14675
+ * deltas (e.g. 'wrote <th' then 'ink> tag'). Escaping each half independently leaves both
14676
+ * halves unescaped, and concatenating them reassembles a literal `<think>`/`</think>` that
14677
+ * is then indistinguishable from the real control marker wrapped around the same text.
14850
14678
  *
14851
- * GPT-Image tiers only - the legacy dall-e path has its own absent-size handling and
14852
- * should gate on isSupportedImageSize directly.
14853
- */
14854
- function resolveGptImageGenerateSize(model, size) {
14855
- const isGptImage2 = isGPTImage2Model(model);
14856
- if (!size) return isGptImage2 ? IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.autoSize : fallbackImageSize(model);
14857
- if (isSupportedImageSize(model, size)) return size;
14858
- if (isGptImage2 && parseSizeEdges(size) === null) return size;
14859
- return fallbackImageSize(model);
14679
+ * This holds back any trailing substring of the buffered text that could still extend into
14680
+ * a marker (up to `<think>`/`</think>`'s length minus one) until the next push resolves it
14681
+ * one way or the other, or flush() is called at the end of the reasoning span.
14682
+ */
14683
+ function createThinkMarkerEscaper() {
14684
+ let pending = "";
14685
+ return {
14686
+ push(chunk) {
14687
+ if (!chunk) return "";
14688
+ const combined = pending + chunk;
14689
+ const holdLength = partialMarkerSuffixLength(combined);
14690
+ const safeLength = combined.length - holdLength;
14691
+ const safe = combined.slice(0, safeLength);
14692
+ pending = combined.slice(safeLength);
14693
+ return escapeThinkMarkers(safe);
14694
+ },
14695
+ flush() {
14696
+ const remaining = pending;
14697
+ pending = "";
14698
+ return escapeThinkMarkers(remaining);
14699
+ }
14700
+ };
14860
14701
  }
14861
14702
  /** Generation was cut off against the output-token ceiling. */
14862
14703
  const TRUNCATED_FINISH_REASON = "max_tokens";
@@ -14884,6 +14725,71 @@ function isEarlyStop(stopReason) {
14884
14725
  return !!stopReason && EARLY_STOP_FINISH_REASONS.has(stopReason);
14885
14726
  }
14886
14727
  /**
14728
+ * Signs the image URLs `web_search` writes into its tool output, so `/api/search-image` (the
14729
+ * same-origin proxy that fetches them server-side) can verify a URL actually came from our own
14730
+ * search-provider results, not from a hostile page's snippet text steering the model into writing
14731
+ * an attacker-controlled URL with exfiltrated data in the query string.
14732
+ *
14733
+ * The signature is a trailing `&b4mSig=<hex>` (or `?b4mSig=<hex>` with no prior query string)
14734
+ * appended by plain string concatenation - never by reparsing the URL through `URLSearchParams`,
14735
+ * which re-serializes every existing param (e.g. turning a literal space into `+`) and would send
14736
+ * a byte-different URL to a host whose own signature (an imgix/S3-presigned link) covers the exact
14737
+ * query string the provider returned. Anchoring the signature to the END of the string, and
14738
+ * requiring it to be the only occurrence, is what makes verification unambiguous: anything a
14739
+ * tamperer appends after signing - including a second `b4mSig` - changes what's left of the anchor
14740
+ * match, which changes the canonical text, which invalidates the signature. There is deliberately
14741
+ * no "read the first/last `b4mSig` param and ignore the rest" step, since that step is exactly
14742
+ * where an earlier version of this file's bypass lived (verify read the URL's first `b4mSig` via
14743
+ * `searchParams.get` while signing canonicalized by deleting *every* `b4mSig`, so a validly-signed
14744
+ * URL with a second, attacker-authored `b4mSig` appended still verified, and was then forwarded to
14745
+ * `safeFetch` with that attacker data still attached).
14746
+ *
14747
+ * The model is already told to copy an `Images:` line's URL verbatim, so no new instruction is
14748
+ * needed for the signature to survive.
14749
+ *
14750
+ * The signature has no expiry: a stored reply, citable, published page, or curation transcript
14751
+ * persists indefinitely, and nothing re-signs a URL on read, so a TTL would make every card
14752
+ * permanently break once it lapsed rather than bound anything meaningful. What actually bounds
14753
+ * a leaked signed URL's usefulness is the same thing that bounds the route otherwise: it's only
14754
+ * reachable at all behind `jwtOnly` auth (apps/client/pages/api/search-image.ts) and is capped by
14755
+ * a per-user rate limit there.
14756
+ */
14757
+ const SIGNATURE_PARAM = "b4mSig";
14758
+ const TRAILING_SIGNATURE_RE = new RegExp(`[?&]${SIGNATURE_PARAM}=([0-9a-f]{32})$`);
14759
+ const KNOWN_PLACEHOLDER_SECRETS = /* @__PURE__ */ new Set([
14760
+ "",
14761
+ "my-secret-placeholder-value",
14762
+ "not-configured"
14763
+ ]);
14764
+ /**
14765
+ * True for an empty or known-placeholder signing secret. Shared so any caller that would
14766
+ * otherwise sign-and-fail (e.g. web_search deciding whether to pay for image results at all)
14767
+ * uses the same definition `verifyImageUrlSignature` fails closed on, rather than a second one
14768
+ * that could drift out of sync.
14769
+ */
14770
+ function isPlaceholderImageSigningSecret(secret) {
14771
+ return KNOWN_PLACEHOLDER_SECRETS.has(secret ?? "");
14772
+ }
14773
+ function computeSignature(canonical, secret) {
14774
+ return createHmac("sha256", secret).update(canonical).digest("hex").slice(0, 32);
14775
+ }
14776
+ /**
14777
+ * Appends expiry + signature query params by string concatenation (never reparses or
14778
+ * re-serializes the existing query string - see the file-level comment). Returns the URL
14779
+ * unchanged if it fails to parse as a URL, or if it already carries a trailing signature.
14780
+ */
14781
+ function signImageUrl(rawUrl, secret) {
14782
+ try {
14783
+ new URL(rawUrl);
14784
+ } catch {
14785
+ return rawUrl;
14786
+ }
14787
+ if (TRAILING_SIGNATURE_RE.test(rawUrl)) return rawUrl;
14788
+ const separator = rawUrl.includes("?") ? "&" : "?";
14789
+ const signature = computeSignature(rawUrl, secret);
14790
+ return `${rawUrl}${separator}${SIGNATURE_PARAM}=${signature}`;
14791
+ }
14792
+ /**
14887
14793
  * promptMeta.functionCalls fields that must never reach a viewer who only holds a
14888
14794
  * "read this conversation" grant - a session share, a live subscription, a bug-report
14889
14795
  * egress to a third party (Slack/email), or a session clone made by a share holder
@@ -14917,6 +14823,13 @@ const OWNER_ONLY_FUNCTION_CALL_FIELDS = ["returnValue", "error"];
14917
14823
  * and a share/subscribe/clone grant authorizes reading the conversation, not re-reading the
14918
14824
  * owner's corpus through it. The sibling `chunkId` is deliberately kept - an opaque id is not
14919
14825
  * content, and `citables[].id` (the file id) is already unredacted beside it.
14826
+ *
14827
+ * `conflictsWith` (#3041) is kept for the same reason, recorded here so the next reader does not
14828
+ * re-derive it: it holds `fabFileId`s of other cited sources, so it adds a RELATIONSHIP between
14829
+ * chips the viewer can already see rather than a slice of the owner's corpus. That holds only while
14830
+ * the field stays ids - a future version carrying the conflicting SENTENCES (the detector has them:
14831
+ * InconsistencyEvidence.excerpt) would be `fullContext`'s class exactly and would have to join the
14832
+ * list below rather than ride along inside this one.
14920
14833
  */
14921
14834
  const OWNER_ONLY_CITABLE_METADATA_FIELDS = ["fullContext"];
14922
14835
  [...OWNER_ONLY_FUNCTION_CALL_FIELDS.map((field) => `promptMeta.functionCalls.${field}`), ...OWNER_ONLY_CITABLE_METADATA_FIELDS.map((field) => `promptMeta.citables.metadata.${field}`)];
@@ -14953,47 +14866,6 @@ z$1.array(triggerWordSchema).max(20, "Up to 20 trigger words allowed.").transfor
14953
14866
  }
14954
14867
  return out;
14955
14868
  });
14956
- /**
14957
- * Serve gate for uploaded FabFiles: hold-until-scanned, fail-closed on ALL mime
14958
- * types, not just images. A file is serveable only once moderation has run to
14959
- * completion on it, REGARDLESS of its declared `mimeType`.
14960
- *
14961
- * Why gate non-images too: `mimeType` is client-declared at upload time and is only
14962
- * corrected by the S3 scan ~1-2s later (see `moderateUploadedFile`'s byte-sniffing). If this
14963
- * gate special-cased "non-images always serveable" based on that same untrusted declared
14964
- * mimeType, a file uploaded as `application/pdf` but actually a PNG (or vice versa) would be
14965
- * served during that window before the sniff/scan ever runs. Gating on `moderationStatus`
14966
- * alone closes that window for every file, image or not.
14967
- *
14968
- * `moderationStatus` semantics:
14969
- * - 'clean' -> serveable
14970
- * - 'pending' | 'scanning' -> NOT serveable (not yet through the scan)
14971
- * - 'blocked' -> NOT serveable (confirmed block / unscannable format)
14972
- * - null | undefined -> NOT serveable (fail-closed; legacy rows are
14973
- * backfilled to 'clean', see backfill-fabfile-moderation-status.ts)
14974
- *
14975
- * Non-image files (PDFs, docs, text, ...) are NOT scanned by Rekognition, but they still
14976
- * pass through `moderateUploadedFile`/`objectCreated`, which resolves them to 'clean'
14977
- * immediately (no image bytes to hold on), so the hold is brief (one S3 event round trip),
14978
- * not an indefinite block.
14979
- */
14980
- function isImageServeable(f) {
14981
- return f.moderationStatus === "clean";
14982
- }
14983
- /**
14984
- * Is this mime type an image? `image/svg+xml` counts, since vision models receive it
14985
- * as an image.
14986
- *
14987
- * The repo has ~50 inline `startsWith('image/')` checks and they disagree on the edges
14988
- * (case, null handling). Only the ones on the attachment pipeline - composer upload,
14989
- * chat context assembly, attachment capability warnings - have been converted here.
14990
- * Icon pickers, avatar validation and resize eligibility still carry their own copies:
14991
- * same question, unrelated subsystems, and folding them in would have made this a
14992
- * repo-wide diff.
14993
- */
14994
- function isImageAttachment(mimeType) {
14995
- return typeof mimeType === "string" && mimeType.toLowerCase().startsWith("image/");
14996
- }
14997
14869
  IMAGE_SIZE_CONSTRAINTS.BFL.minWidth, IMAGE_SIZE_CONSTRAINTS.BFL.maxWidth;
14998
14870
  Array.from(new Set([
14999
14871
  {
@@ -15901,71 +15773,10 @@ Array.from(new Set([
15901
15773
  const [, top] = v.target.split("/");
15902
15774
  return `/${top}`;
15903
15775
  })));
15904
- function getHeader(headers, name) {
15905
- if (!headers || typeof headers !== "object") return null;
15906
- if (typeof headers.get === "function") {
15907
- const value = headers.get(name);
15908
- return typeof value === "string" ? value : null;
15909
- }
15910
- const value = headers[name] ?? headers[name.toLowerCase()];
15911
- return typeof value === "string" ? value : null;
15912
- }
15913
- function parseCount(value) {
15914
- if (value === null) return null;
15915
- const trimmed = value.trim();
15916
- if (!trimmed) return null;
15917
- const parsed = Number(trimmed);
15918
- return Number.isFinite(parsed) ? parsed : null;
15919
- }
15920
- const UNIT_MS = {
15921
- ms: 1,
15922
- s: 1e3,
15923
- m: 6e4,
15924
- h: 36e5
15925
- };
15926
- const DURATION_PART = /(\d+(?:\.\d+)?)(ms|h|m|s)/g;
15927
- /**
15928
- * Parse a Go-style duration ("6ms", "0s", "1m30s", "1h2m3s") to milliseconds.
15929
- *
15930
- * Exported for its own tests: it is the part of this module that can be wrong in a way the
15931
- * numbers still look plausible.
15932
- */
15933
- function parseDurationMs(value) {
15934
- if (typeof value !== "string") return null;
15935
- const trimmed = value.trim();
15936
- if (!trimmed) return null;
15937
- DURATION_PART.lastIndex = 0;
15938
- let total = 0;
15939
- let matched = 0;
15940
- let consumed = 0;
15941
- for (const part of trimmed.matchAll(DURATION_PART)) {
15942
- total += Number(part[1]) * UNIT_MS[part[2]];
15943
- consumed += part[0].length;
15944
- matched += 1;
15945
- }
15946
- if (matched === 0 || consumed !== trimmed.length) return null;
15947
- return total;
15948
- }
15949
- /** Read both rate-limit dimensions off a provider response. */
15950
- function parseEmbeddingRateLimitHeaders(headers) {
15951
- return {
15952
- limitTokens: parseCount(getHeader(headers, "x-ratelimit-limit-tokens")),
15953
- limitRequests: parseCount(getHeader(headers, "x-ratelimit-limit-requests")),
15954
- remainingTokens: parseCount(getHeader(headers, "x-ratelimit-remaining-tokens")),
15955
- remainingRequests: parseCount(getHeader(headers, "x-ratelimit-remaining-requests")),
15956
- resetTokensMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-tokens")),
15957
- resetRequestsMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-requests"))
15958
- };
15959
- }
15960
- /** True when the provider reported at least one usable ceiling. */
15961
- function hasUsableLimits(snapshot) {
15962
- return snapshot.limitTokens !== null || snapshot.limitRequests !== null;
15963
- }
15964
15776
  dayjs.extend(utc);
15965
15777
  dayjs.extend(timezone);
15966
15778
  dayjs.extend(relativeTime);
15967
15779
  dayjs.extend(localizedFormat);
15968
- var dayjsConfig_default = dayjs;
15969
15780
  /**
15970
15781
  * Default retryable errors for LLM API calls
15971
15782
  */
@@ -16133,42 +15944,6 @@ async function withRetry(fn, options = {}) {
16133
15944
  }
16134
15945
  }
16135
15946
  }
16136
- function readZipEntryBounded(entry, maxBytes) {
16137
- return new Promise((resolve, reject) => {
16138
- const stream = entry.internalStream("nodebuffer");
16139
- let parts = [];
16140
- let byteLength = 0;
16141
- let settled = false;
16142
- stream.on("data", (chunk) => {
16143
- if (settled) return;
16144
- byteLength += chunk.length;
16145
- if (byteLength > maxBytes) {
16146
- settled = true;
16147
- stream.pause();
16148
- parts = [];
16149
- resolve({
16150
- ok: false,
16151
- reason: "too-large"
16152
- });
16153
- return;
16154
- }
16155
- parts.push(chunk);
16156
- }).on("error", (error) => {
16157
- if (settled) return;
16158
- settled = true;
16159
- parts = [];
16160
- reject(error);
16161
- }).on("end", () => {
16162
- if (settled) return;
16163
- settled = true;
16164
- resolve({
16165
- ok: true,
16166
- text: Buffer.concat(parts).toString("utf8"),
16167
- byteLength
16168
- });
16169
- }).resume();
16170
- });
16171
- }
16172
15947
  //#endregion
16173
15948
  //#region src/utils/apiUrl.ts
16174
15949
  /**
@@ -16290,6 +16065,19 @@ function getEnvironmentName(configApiConfig) {
16290
16065
  return "Self-Hosted";
16291
16066
  }
16292
16067
  //#endregion
16068
+ //#region src/utils/validateSessionId.ts
16069
+ /**
16070
+ * Session and resume ids arrive from the environment (`B4M_SESSION_ID`,
16071
+ * `B4M_RESUME_ID`) and are used as filesystem path components by the session
16072
+ * store and the debug logger. Restrict them to a strict charset (which still
16073
+ * covers UUIDs) so a hostile launcher cannot traverse out of the base dir via
16074
+ * e.g. `B4M_SESSION_ID=../config`.
16075
+ */
16076
+ const SESSION_ID_PATTERN = /^[A-Za-z0-9_-]+$/;
16077
+ function isValidSessionId(value) {
16078
+ return SESSION_ID_PATTERN.test(value);
16079
+ }
16080
+ //#endregion
16293
16081
  //#region src/config/toolSafety.ts
16294
16082
  /**
16295
16083
  * Tool safety categories determine when permission is required
@@ -16405,13 +16193,13 @@ function formatFileSize(bytes) {
16405
16193
  function tryReadContextFile(dir, filename, source) {
16406
16194
  const filePath = path$1.join(dir, filename);
16407
16195
  try {
16408
- const stats = fs$2.lstatSync(filePath);
16196
+ const stats = fs$1.lstatSync(filePath);
16409
16197
  if (stats.isDirectory()) return null;
16410
16198
  if (stats.isSymbolicLink()) return { error: `${source === "global" ? "Global" : "Project"} ${filename} is a symlink (not allowed for security)` };
16411
16199
  if (stats.size > 102400) return { error: `${source === "global" ? "Global" : "Project"} ${filename} exceeds 100KB limit (${formatFileSize(stats.size)})` };
16412
16200
  return {
16413
16201
  filename,
16414
- content: fs$2.readFileSync(filePath, "utf-8"),
16202
+ content: fs$1.readFileSync(filePath, "utf-8"),
16415
16203
  source,
16416
16204
  path: filePath
16417
16205
  };
@@ -16513,9 +16301,10 @@ const logger = class Logger {
16513
16301
  * Initialize the logger with a session ID
16514
16302
  */
16515
16303
  async initialize(sessionId) {
16304
+ if (!isValidSessionId(sessionId)) throw new Error(`Invalid session id "${sessionId}": must match ${SESSION_ID_PATTERN.source}`);
16516
16305
  this.sessionId = sessionId;
16517
16306
  const debugDir = path.join(os.homedir(), ".bike4mind", "debug");
16518
- await fs$1.mkdir(debugDir, { recursive: true });
16307
+ await fs.mkdir(debugDir, { recursive: true });
16519
16308
  this.logFilePath = path.join(debugDir, `${sessionId}.txt`);
16520
16309
  await this.writeToFile("INFO", "=== CLI SESSION START ===");
16521
16310
  }
@@ -16567,7 +16356,7 @@ const logger = class Logger {
16567
16356
  if (!this.fileLoggingEnabled || !this.logFilePath) return;
16568
16357
  try {
16569
16358
  const logEntry = `[${(/* @__PURE__ */ new Date()).toISOString().replace("T", " ").substring(0, 19)}] [${level}] ${message}\n`;
16570
- await fs$1.appendFile(this.logFilePath, logEntry, "utf-8");
16359
+ await fs.appendFile(this.logFilePath, logEntry, "utf-8");
16571
16360
  } catch (error) {
16572
16361
  console.error("File logging failed:", error);
16573
16362
  }
@@ -16684,11 +16473,11 @@ const logger = class Logger {
16684
16473
  if (!this.fileLoggingEnabled) return;
16685
16474
  try {
16686
16475
  const debugDir = path.join(os.homedir(), ".bike4mind", "debug");
16687
- const files = await fs$1.readdir(debugDir);
16476
+ const files = await fs.readdir(debugDir);
16688
16477
  const thirtyDaysAgo = Date.now() - 2592e6;
16689
16478
  for (const file of files) {
16690
16479
  const filePath = path.join(debugDir, file);
16691
- if ((await fs$1.stat(filePath)).mtime.getTime() < thirtyDaysAgo) await fs$1.unlink(filePath);
16480
+ if ((await fs.stat(filePath)).mtime.getTime() < thirtyDaysAgo) await fs.unlink(filePath);
16692
16481
  }
16693
16482
  } catch (error) {
16694
16483
  console.error("Failed to cleanup old logs:", error);
@@ -17640,7 +17429,7 @@ var ConfigStore = class {
17640
17429
  */
17641
17430
  async saveSandboxConfig(sandbox) {
17642
17431
  await this.load();
17643
- this.globalConfig.sandbox = sandbox;
17432
+ this.globalConfig.sandbox = structuredClone(sandbox);
17644
17433
  await this.save();
17645
17434
  }
17646
17435
  /**
@@ -17931,4 +17720,4 @@ var ConfigStore = class {
17931
17720
  }
17932
17721
  };
17933
17722
  //#endregion
17934
- export { RESPONSES_API_TOOL_MODELS as $, secureParameters as $t, DEFAULT_UNKNOWN_CONTEXT_WINDOW as A, isGPTImage2Model as At, InternalServerError as B, isRetryableError as Bt, BadRequestError as C, hasUsableLimits as Ct, ChatModels as D, isChunkStalledFile as Dt, CREDIT_DEDUCT_TRANSACTION_TYPES as E, isChunkRebuildPending as Et, ForbiddenError as F, isMediaModelType as Ft, NotFoundError as G, isZodError as Gt, McpServerName as H, isSupportedImageSize as Ht, HTTPError as I, isModelAccessible as It, PROMPT_TEXT_MAX as J, parseEmbeddingRateLimitHeaders as Jt, OllamaEmbeddingModel as K, mapMimeTypeToArtifactType as Kt, HttpStatus as L, isModelDeprecated as Lt, FIELD_GROUP_OF as M, isGeminiModelId as Mt, FIXED_TEMPERATURE_MODELS as N, isImageAttachment as Nt, CorruptedFileError as O, isEarlyStop as Ot, FORMAT_PROMPT_TEMPLATE as P, isImageServeable as Pt, REFUSAL_FALLBACK_MODELS as Q, resolveHistoryFetchLimit as Qt, IMAGE_SIZE_CONSTRAINTS as R, isPlaceholderApiKey as Rt, BFL_SAFETY_TOLERANCE as S, hasKeylessCloudEmbedder as St, CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS as T, isBflUltraImageModel as Tt, ModelBackend as U, isUnlimitedHistory as Ut, MODEL_INFO_FIELD_GROUP_OF as V, isSupportedFabFileMimeType as Vt, NO_TEMPERATURE_MODELS as W, isUserInitiatedAbort as Wt, REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS as X, reservationOutputTokens as Xt, PermissionDeniedError as Y, readZipEntryBounded as Yt, REASONING_SUPPORTED_MODELS as Z, resolveGptImageGenerateSize as Zt, AGENT_QUEST_MCP_URI as _, defaultEmbeddingModelForEnv as _t, canTrustTool as a, usdToCreditsStochastic as an, TooManyRequestsError as at, AudioMimeType as b, getQuestErrorCode as bt, ApiEndpointUnconfiguredError as c, actorColorIndex as cn, VIDEO_SIZE_CONSTRAINTS as ct, getEnvironmentName as d, selfClaimedActorKindSchema as dn, WORK_ITEM_STATUSES as dt, settingsMap as en, REVIEW_GATE_STATUS_VALUES as et, parseApiUrl as f, buildRateLimitLogEntry as fn, applyModelPriceCatalog as ft, AGENT_QUEST_MANIFEST as g, dayjsConfig_default as gt, AGENT_QUEST_ID as h, parseRateLimitHeaders as hn, countCodePoints as ht, loadContextFiles as i, usdToCredits as in, TTS_MAX_INPUT_CHARS as it, DEGENERATE_FINISH_REASON as j, isGPTImageModel as jt, DEFAULT_MUSIC_MODEL_ID as k, isFieldGroup as kt, LOCAL_DEV_URL as l, actorKindMarker as ln, VideoModels as lt, resolveApiEndpoint as m, isNearLimit as mn, capForParse as mt, logger as n, toModelRecord as nn, SpeechToTextModels as nt, getToolCategory as o, withRetry as on, UnauthorizedError as ot, requireApiUrl as p, extractSnippetMeta as pn, calculateRetryDelay as pt, OpenAIEmbeddingModel as q, obfuscateApiKey as qt, extractCompactInstructions as r, toNonWebpOutputFormat as rn, SupportedFabFileMimeTypes as rt, isReadOnlyTool as s, ACTOR_COLOR_SLOTS as sn, UnprocessableEntityError as st, ConfigStore as t, toModelInfo as tn, SUBQUEST_STATUS_VALUES as tt, getCreditsUrl as u, actorKindSchema as un, VoyageAIEmbeddingModel as ut, ARTIFACT_ATTRS_PATTERN as v, fallbackImageSize as vt, BedrockEmbeddingModel as w, isAudioMimeType as wt, BEDROCK_NO_PROMPT_CACHING_MODELS as x, getRetryAfterMs as xt, ApiKeyType as y, getMcpProviderMetadata as yt, ImageModels as z, isRenderableModelType as zt };
17723
+ export { withRetry as $, NotFoundError as A, WORK_ITEM_STATUSES as B, CREDIT_DEDUCT_TRANSACTION_TYPES as C, MODEL_INFO_FIELD_GROUP_OF as D, LOCATION_MAP_LANGUAGE as E, REVIEW_GATE_STATUS_VALUES as F, googleMapsSearchUrl as G, escapeThinkMarkers as H, SEARCH_RESULT_CARDS_LANGUAGE as I, isRetryableError as J, isEarlyStop as K, SUBQUEST_STATUS_VALUES as L, OpenAIEmbeddingModel as M, PROMPT_TEXT_MAX as N, McpServerName as O, PermissionDeniedError as P, signImageUrl as Q, SupportedFabFileMimeTypes as R, CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS as S, DEGENERATE_FINISH_REASON as T, getMcpProviderMetadata as U, createThinkMarkerEscaper as V, getQuestErrorCode as W, obfuscateApiKey as X, isUserInitiatedAbort as Y, secureParameters as Z, AGENT_QUEST_ID as _, canTrustTool as a, ApiKeyType as b, SESSION_ID_PATTERN as c, LOCAL_DEV_URL as d, ACTOR_COLOR_SLOTS as et, getCreditsUrl as f, resolveApiEndpoint as g, requireApiUrl as h, loadContextFiles as i, selfClaimedActorKindSchema as it, OllamaEmbeddingModel as j, ModelBackend as k, isValidSessionId as l, parseApiUrl as m, logger as n, actorKindMarker as nt, getToolCategory as o, getEnvironmentName as p, isPlaceholderImageSigningSecret as q, extractCompactInstructions as r, actorKindSchema as rt, isReadOnlyTool as s, ConfigStore as t, actorColorIndex as tt, ApiEndpointUnconfiguredError as u, AGENT_QUEST_MANIFEST as v, ChatModels as w, BedrockEmbeddingModel as x, AGENT_QUEST_MCP_URI as y, VoyageAIEmbeddingModel as z };