@bike4mind/cli 0.20.2 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,7 +7,6 @@ import path from "path";
7
7
  import { v4 } from "uuid";
8
8
  import * as z$2 from "zod";
9
9
  import z, { ZodError, z as z$1 } from "zod";
10
- import { actorKindSchema, hearthEventKindSchema, hearthEventRefsSchema, hearthMachineBodySchema } from "@bike4mind/hearth";
11
10
  import dayjs from "dayjs";
12
11
  import timezone from "dayjs/plugin/timezone.js";
13
12
  import utc from "dayjs/plugin/utc.js";
@@ -52,7 +51,7 @@ const extractSnippetMeta = (content) => {
52
51
  }
53
52
  return { sections };
54
53
  };
55
- function getHeader(headers, name) {
54
+ function getHeader$1(headers, name) {
56
55
  if (!headers || typeof headers !== "object") return null;
57
56
  if (typeof headers.get === "function") {
58
57
  const val = headers.get(name);
@@ -76,10 +75,10 @@ function parseNumber(value) {
76
75
  * - `Retry-After` - seconds to wait (on 429 responses), or an HTTP-date
77
76
  */
78
77
  function parseRateLimitHeaders(headers) {
79
- const limitStr = getHeader(headers, "X-RateLimit-Limit") ?? getHeader(headers, "x-ratelimit-limit");
80
- const remainingStr = getHeader(headers, "X-RateLimit-Remaining") ?? getHeader(headers, "x-ratelimit-remaining");
81
- const resetStr = getHeader(headers, "X-RateLimit-Reset") ?? getHeader(headers, "x-ratelimit-reset");
82
- const retryAfterStr = getHeader(headers, "Retry-After") ?? getHeader(headers, "retry-after");
78
+ const limitStr = getHeader$1(headers, "X-RateLimit-Limit") ?? getHeader$1(headers, "x-ratelimit-limit");
79
+ const remainingStr = getHeader$1(headers, "X-RateLimit-Remaining") ?? getHeader$1(headers, "x-ratelimit-remaining");
80
+ const resetStr = getHeader$1(headers, "X-RateLimit-Reset") ?? getHeader$1(headers, "x-ratelimit-reset");
81
+ const retryAfterStr = getHeader$1(headers, "Retry-After") ?? getHeader$1(headers, "retry-after");
83
82
  const limit = parseNumber(limitStr);
84
83
  const remaining = parseNumber(remainingStr);
85
84
  let resetAt = null;
@@ -110,13 +109,6 @@ function parseRateLimitHeaders(headers) {
110
109
  usagePercent
111
110
  };
112
111
  }
113
- /**
114
- * Check whether the current rate limit usage is near the threshold.
115
- *
116
- * @param info - Parsed rate limit info
117
- * @param thresholdPercent - Usage percentage threshold (default: 80)
118
- * @returns true if usage is at or above the threshold
119
- */
120
112
  function isNearLimit(info, thresholdPercent = 80) {
121
113
  if (info.usagePercent === null) return false;
122
114
  return info.usagePercent >= thresholdPercent;
@@ -140,6 +132,151 @@ function buildRateLimitLogEntry(integration, endpoint, info, wasThrottled = fals
140
132
  };
141
133
  }
142
134
  //#endregion
135
+ //#region ../../b4m-core/hearth/dist/index.mjs
136
+ /**
137
+ * Zod validation for data crossing the Hearth boundary (API routes, CLI
138
+ * tools, gateways). Must stay in sync with the types in types.ts.
139
+ */
140
+ const actorKindSchema = z$1.enum([
141
+ "human",
142
+ "agent",
143
+ "gateway",
144
+ "device",
145
+ "system"
146
+ ]);
147
+ /**
148
+ * The kinds a caller may claim FOR ITSELF. 'human' and 'system' are reserved:
149
+ * the human actor is derived from the authenticated session, never from a
150
+ * request body, so no credential can post an event that renders as the account
151
+ * owner. Claiming one of these three is a downgrade in trust, not a spoof.
152
+ *
153
+ * Single source for both self-identification paths - the actor override and the
154
+ * per-session kind (see HearthActorParamSchema / HearthSessionParamSchema in
155
+ * the client's hearthWire) - so the reserved set cannot drift between them.
156
+ */
157
+ const selfClaimedActorKindSchema = z$1.enum([
158
+ "agent",
159
+ "gateway",
160
+ "device"
161
+ ]);
162
+ const hearthEventKindSchema = z$1.enum([
163
+ "message",
164
+ "edit",
165
+ "reaction",
166
+ "artifact",
167
+ "presence",
168
+ "delegation",
169
+ "quest.update",
170
+ "gate.request",
171
+ "gate.resolve",
172
+ "system"
173
+ ]);
174
+ const hearthHumanBodySchema = z$1.object({
175
+ text: z$1.string().min(1),
176
+ format: z$1.enum(["md", "text"])
177
+ });
178
+ const hearthMachineBodySchema = z$1.object({
179
+ schema: z$1.string().min(1),
180
+ payload: z$1.unknown()
181
+ });
182
+ const hearthEventRefsSchema = z$1.object({
183
+ threadRootId: z$1.string().min(1).optional(),
184
+ replyToId: z$1.string().min(1).optional(),
185
+ questId: z$1.string().min(1).optional(),
186
+ externalId: z$1.string().min(1).optional()
187
+ });
188
+ z$1.object({
189
+ channelId: z$1.string().min(1),
190
+ actorId: z$1.string().min(1),
191
+ kind: hearthEventKindSchema,
192
+ human: hearthHumanBodySchema,
193
+ machine: hearthMachineBodySchema.optional(),
194
+ refs: hearthEventRefsSchema
195
+ });
196
+ /** djb2. Deterministic across processes and restarts, which is the whole point. */
197
+ function hashOf(value) {
198
+ let hash = 5381;
199
+ for (let i = 0; i < value.length; i++) hash = (hash << 5) + hash + value.charCodeAt(i) | 0;
200
+ return Math.abs(hash);
201
+ }
202
+ /**
203
+ * Stable palette slot for an actor. Hash-derived, never array index or arrival
204
+ * order: those repaint every actor whenever the tail changes or a reload
205
+ * reorders the buffer, and a shifting color is worse than no color for telling
206
+ * two agents apart.
207
+ */
208
+ function actorColorIndex(actorId) {
209
+ if (!actorId) return 0;
210
+ return hashOf(actorId) % 4;
211
+ }
212
+ const ACTOR_COLOR_SLOTS = [
213
+ {
214
+ light: "#2a78d6",
215
+ dark: "#3987e5"
216
+ },
217
+ {
218
+ light: "#eda100",
219
+ dark: "#c98500"
220
+ },
221
+ {
222
+ light: "#e87ba4",
223
+ dark: "#d55181"
224
+ },
225
+ {
226
+ light: "#008300",
227
+ dark: "#008300"
228
+ }
229
+ ];
230
+ /** Single-character marker for text-only surfaces (the CLI /hearth listing). */
231
+ const ACTOR_KIND_MARKERS = {
232
+ human: "H",
233
+ agent: "A",
234
+ gateway: "G",
235
+ device: "D",
236
+ system: "S"
237
+ };
238
+ /**
239
+ * Read through a Partial view on purpose. The type is closed within one build,
240
+ * but the SPA bare-casts the WS payload, so a client running stale code against
241
+ * a server that has added a kind receives one that is absent from these maps -
242
+ * and a blank badge reads as "no kind stated" rather than "kind unknown", which
243
+ * is the opposite of what the mitigation needs to say.
244
+ */
245
+ function lookup(map, kind, fallback) {
246
+ if (!kind) return fallback;
247
+ return map[kind] ?? fallback;
248
+ }
249
+ function actorKindMarker(kind) {
250
+ return lookup(ACTOR_KIND_MARKERS, kind, "?");
251
+ }
252
+ z$1.object({
253
+ /** Claude Code lifecycle event name; the reason fallback for tiers 0 and 1. */
254
+ hook_event_name: z$1.string().nullish(),
255
+ session_id: z$1.string().nullish(),
256
+ slug: z$1.string().nullish(),
257
+ /** Workspace BASENAME. Never a full path - that is content. */
258
+ workspace: z$1.string().nullish(),
259
+ /**
260
+ * Which reporter wrote this. A LOOSE string, not an enum: a fourth surface
261
+ * posting an unrecognized value must land on the roster with its detail
262
+ * intact, and a strict enum would fail the whole parse and drop the row - the
263
+ * exact "a third reporter is expensive" cost this contract exists to remove.
264
+ */
265
+ surface: z$1.string().nullish(),
266
+ /** Engine driving the session, where the reporter knows it. */
267
+ source: z$1.string().nullish(),
268
+ claude_version: z$1.string().nullish(),
269
+ activity: z$1.object({
270
+ reason: z$1.string().nullish(),
271
+ tool: z$1.string().nullish(),
272
+ permission_mode: z$1.string().nullish(),
273
+ effort: z$1.string().nullish(),
274
+ duration_ms: z$1.number().nullish(),
275
+ subagent: z$1.string().nullish(),
276
+ background_tasks: z$1.number().int().min(0).nullish()
277
+ }).nullish()
278
+ });
279
+ //#endregion
143
280
  //#region ../../b4m-core/common/dist/index.mjs
144
281
  let HttpStatus = /* @__PURE__ */ function(HttpStatus) {
145
282
  HttpStatus[HttpStatus["Ok"] = 200] = "Ok";
@@ -148,9 +285,11 @@ let HttpStatus = /* @__PURE__ */ function(HttpStatus) {
148
285
  HttpStatus[HttpStatus["Unauthorized"] = 401] = "Unauthorized";
149
286
  HttpStatus[HttpStatus["Forbidden"] = 403] = "Forbidden";
150
287
  HttpStatus[HttpStatus["NotFound"] = 404] = "NotFound";
288
+ HttpStatus[HttpStatus["Conflict"] = 409] = "Conflict";
151
289
  HttpStatus[HttpStatus["UnprocessableEntity"] = 422] = "UnprocessableEntity";
152
290
  HttpStatus[HttpStatus["TooManyRequests"] = 429] = "TooManyRequests";
153
291
  HttpStatus[HttpStatus["InternalServerError"] = 500] = "InternalServerError";
292
+ HttpStatus[HttpStatus["BadGateway"] = 502] = "BadGateway";
154
293
  return HttpStatus;
155
294
  }({});
156
295
  var HTTPError = class extends Error {
@@ -277,6 +416,7 @@ let ApiKeyType = /* @__PURE__ */ function(ApiKeyType) {
277
416
  ApiKeyType["ollama"] = "ollama";
278
417
  ApiKeyType["xai"] = "xai";
279
418
  ApiKeyType["kimi"] = "kimi";
419
+ ApiKeyType["deepseek"] = "deepseek";
280
420
  ApiKeyType["voyageai"] = "voyageai";
281
421
  return ApiKeyType;
282
422
  }({});
@@ -312,6 +452,7 @@ const b4mLLMTools = z$1.enum([
312
452
  "chess_engine",
313
453
  "retrieve_knowledge_content",
314
454
  "count_knowledge_base",
455
+ "describe_knowledge_base",
315
456
  "delegate_to_agent",
316
457
  "optihashi_schedule",
317
458
  "optihashi_formulate",
@@ -349,6 +490,16 @@ const FallbackInfoSchema = z$1.object({
349
490
  primaryModelName: z$1.string(),
350
491
  fallbackModel: z$1.string(),
351
492
  fallbackModelName: z$1.string(),
493
+ /**
494
+ * Backend of each side, so the badge can name the provider path without
495
+ * inferring it from the id - a bare `deepseek-*` slug is the direct API on a
496
+ * hosted deployment and an Ollama pull on a self-hosted one. Optional and
497
+ * loosely typed: these payloads are persisted in browser storage, so a record
498
+ * written by an older build has neither field and an unknown backend name from
499
+ * a newer one must not fail the parse.
500
+ */
501
+ primaryModelBackend: z$1.string().optional(),
502
+ fallbackModelBackend: z$1.string().optional(),
352
503
  timestamp: z$1.number()
353
504
  });
354
505
  const ArtifactMetadataSchema = z$1.object({
@@ -755,6 +906,7 @@ let ModelBackend = /* @__PURE__ */ function(ModelBackend) {
755
906
  ModelBackend["BFL"] = "bfl";
756
907
  ModelBackend["XAI"] = "xai";
757
908
  ModelBackend["Kimi"] = "kimi";
909
+ ModelBackend["DeepSeek"] = "deepseek";
758
910
  ModelBackend["VoyageAI"] = "voyageai";
759
911
  ModelBackend["AWS"] = "aws";
760
912
  ModelBackend["LocalImage"] = "local-image";
@@ -785,6 +937,62 @@ let ImageModels = /* @__PURE__ */ function(ImageModels) {
785
937
  Object.values(ImageModels);
786
938
  z$1.enum(ImageModels);
787
939
  /**
940
+ * Image size constraints and options
941
+ */
942
+ const IMAGE_SIZE_CONSTRAINTS = {
943
+ BFL: {
944
+ minWidth: 256,
945
+ maxWidth: 1440,
946
+ minHeight: 256,
947
+ maxHeight: 1440,
948
+ stepSize: 32,
949
+ defaultSize: "1280x960",
950
+ sizes: [
951
+ "1280x960",
952
+ "1024x768",
953
+ "800x600",
954
+ "1280x720",
955
+ "1024x576",
956
+ "1440x810",
957
+ "1024x1024",
958
+ "768x768",
959
+ "512x512",
960
+ "960x1280",
961
+ "768x1024",
962
+ "600x800"
963
+ ]
964
+ },
965
+ GPT_IMAGE_1: {
966
+ sizes: [
967
+ "1024x1024",
968
+ "1024x1536",
969
+ "1536x1024"
970
+ ],
971
+ defaultSize: "1024x1024"
972
+ },
973
+ GPT_IMAGE_2: {
974
+ /** Popular preset sizes shown in the UI. The API accepts any resolution meeting the constraints. */
975
+ sizes: [
976
+ "1024x1024",
977
+ "1536x1024",
978
+ "1024x1536",
979
+ "2048x2048",
980
+ "2048x1152",
981
+ "3840x2160",
982
+ "2160x3840"
983
+ ],
984
+ defaultSize: "1024x1024",
985
+ /** Constraints for custom/flexible sizes */
986
+ constraints: {
987
+ maxEdge: 3840,
988
+ minTotalPixels: 655360,
989
+ maxTotalPixels: 8294400,
990
+ edgeMultiple: 16,
991
+ maxAspectRatio: 3
992
+ }
993
+ }
994
+ };
995
+ /**
788
996
  * Chat Models
789
997
  *
790
998
  * https://platform.openai.com/docs/models/continuous-model-upgrades
@@ -897,6 +1105,7 @@ let ChatModels = /* @__PURE__ */ function(ChatModels) {
897
1105
  ChatModels["KIMI_K2_5"] = "kimi-k2.5";
898
1106
  ChatModels["KIMI_K2_5_BEDROCK"] = "moonshotai.kimi-k2.5";
899
1107
  ChatModels["KIMI_K2_THINKING_BEDROCK"] = "moonshot.kimi-k2-thinking";
1108
+ ChatModels["DEEPSEEK_FLASH"] = "deepseek-flash";
900
1109
  return ChatModels;
901
1110
  }({});
902
1111
  const CHAT_MODELS = Object.values(ChatModels);
@@ -1031,7 +1240,8 @@ const NO_TEMPERATURE_MODELS = /* @__PURE__ */ new Set([
1031
1240
  "kimi-k2.7-code",
1032
1241
  "kimi-k2.7-code-highspeed",
1033
1242
  "kimi-k2.6",
1034
- "kimi-k2.5"
1243
+ "kimi-k2.5",
1244
+ "deepseek-flash"
1035
1245
  ]);
1036
1246
  /**
1037
1247
  * Models whose safety classifiers can decline a request with `stop_reason: 'refusal'`
@@ -1122,6 +1332,14 @@ const SUBQUEST_STATUS_VALUES = [
1122
1332
  "deleted"
1123
1333
  ];
1124
1334
  /**
1335
+ * Status of a review gate on a sub-quest
1336
+ */
1337
+ const REVIEW_GATE_STATUS_VALUES = [
1338
+ "pending",
1339
+ "approved",
1340
+ "rejected"
1341
+ ];
1342
+ /**
1125
1343
  * Valid complexity ratings for quests.
1126
1344
  * Canonical vocabulary - matches what the planner generates and validates
1127
1345
  * (Easy < 1 hour, Medium 1-4 hours, Hard > 4 hours).
@@ -1512,6 +1730,7 @@ const ADAPTER_FAMILIES = [
1512
1730
  "gemini",
1513
1731
  "xai",
1514
1732
  "kimi",
1733
+ "deepseek",
1515
1734
  "ollama",
1516
1735
  "bfl",
1517
1736
  "local-image",
@@ -1762,6 +1981,7 @@ const MODEL_INFO_FIELD_GROUP_OF = {
1762
1981
  supportsImageVariation: "modalities",
1763
1982
  supportsSafetyTolerance: "modalities",
1764
1983
  deprecationDate: "lifecycle",
1984
+ replacedBy: "lifecycle",
1765
1985
  adapterFamily: "dispatch",
1766
1986
  dispatchProfile: "dispatch",
1767
1987
  description: "presentation",
@@ -1885,6 +2105,8 @@ z$1.object({
1885
2105
  }).optional(),
1886
2106
  /** Optional: a state row written before Phase 4 has none, and must keep parsing. */
1887
2107
  suggestion: ModelLifecycleSuggestion.optional(),
2108
+ /** Dispatch probes spent on this model, so one that always fails stops taking a budget slot. */
2109
+ probeAttempts: z$1.number().int().nonnegative().optional(),
1888
2110
  createdAt: z$1.date(),
1889
2111
  updatedAt: z$1.date()
1890
2112
  });
@@ -2188,6 +2410,13 @@ z$1.object({
2188
2410
  createdAt: z$1.date(),
2189
2411
  updatedAt: z$1.date()
2190
2412
  });
2413
+ let McpServerName = /* @__PURE__ */ function(McpServerName) {
2414
+ McpServerName["LinkedIn"] = "linkedin";
2415
+ McpServerName["Github"] = "github";
2416
+ McpServerName["Atlassian"] = "atlassian";
2417
+ McpServerName["Notion"] = "notion";
2418
+ return McpServerName;
2419
+ }({});
2191
2420
  /**
2192
2421
  * Check if a value is a placeholder (not configured or SST default).
2193
2422
  * Uses case-insensitive comparison and trims whitespace to prevent bypass attempts.
@@ -2233,6 +2462,42 @@ function isPlaceholderApiKey(value) {
2233
2462
  if (!normalized) return true;
2234
2463
  return PLACEHOLDER_API_KEY_REGEX.test(normalized);
2235
2464
  }
2465
+ /**
2466
+ * Lake lifecycle. Stable states (draft/active/archived/deleted) plus transitional
2467
+ * states (archiving/unarchiving/restoring/deleting/purging) that exist to drive UI and make a crashed
2468
+ * mid-operation observable. draft -> active is one-way. It happens implicitly once the lake
2469
+ * holds its first member file (see `activateIfDraft` below), and unconditionally when an
2470
+ * archived or deleted lake is restored, which is how an empty lake can end up active.
2471
+ *
2472
+ * `purging` is the one transitional state that is NOT recoverable by retrying the same action:
2473
+ * it is claimed the moment a phase-2 hard delete is ACCEPTED (#1744), before the background
2474
+ * sweep runs, so that `listDeletedDataLakes` stops offering Restore on a lake whose
2475
+ * destruction is already irreversible. Everything else that reads `status` must treat it as
2476
+ * "going away", never as a lake to act on.
2477
+ */
2478
+ const DATA_LAKE_STATUSES = [
2479
+ "draft",
2480
+ "active",
2481
+ "archiving",
2482
+ "archived",
2483
+ "unarchiving",
2484
+ "restoring",
2485
+ "deleting",
2486
+ "deleted",
2487
+ "purging"
2488
+ ];
2489
+ /**
2490
+ * Stable (non-transitional) lake statuses - a lake sitting in one of these is at rest, not
2491
+ * mid-operation. Load-bearing as the INPUT to `DATA_LAKE_TRANSITIONAL_STATUSES` below, which is
2492
+ * what drives the needs-attention list; it is not itself a filter any list path applies.
2493
+ */
2494
+ const DATA_LAKE_STABLE_STATUSES = [
2495
+ "draft",
2496
+ "active",
2497
+ "archived",
2498
+ "deleted"
2499
+ ];
2500
+ DATA_LAKE_STATUSES.filter((s) => !DATA_LAKE_STABLE_STATUSES.includes(s));
2236
2501
  z$1.object({
2237
2502
  /**
2238
2503
  * The granting lake's Mongo `_id`. ALWAYS a persisted DB lake: a hardcoded/fallback lake has no
@@ -2263,6 +2528,23 @@ z$1.object({
2263
2528
  */
2264
2529
  expiresAt: z$1.date().nullish()
2265
2530
  });
2531
+ /** One prefix-arm content tag a lake held on a file, as captured at removal for later restore. */
2532
+ const LakeMembershipRemovalContentTag = z$1.object({
2533
+ name: z$1.string(),
2534
+ strength: z$1.number()
2535
+ });
2536
+ z$1.object({
2537
+ dataLakeId: z$1.string(),
2538
+ fabFileId: z$1.string(),
2539
+ /** The principal who performed the removal - audit only, never a restore authorization input. */
2540
+ actorUserId: z$1.string(),
2541
+ /** The lake's prefix-arm tags the file carried, captured by `lakeMembershipSignals`. May be
2542
+ * empty - a meta-tag-only member and a non-creator-owned file both removed with no content
2543
+ * tags to restore, which is a complete and legitimate answer, not "nothing to record". */
2544
+ contentTags: z$1.array(LakeMembershipRemovalContentTag),
2545
+ removedAt: z$1.date(),
2546
+ expiresAt: z$1.date()
2547
+ });
2266
2548
  /**
2267
2549
  * Every `IDataLake` field, classified as audited or not. A TOTAL map keyed by `keyof IDataLake`,
2268
2550
  * exactly like `LAKE_FIELD_VISIBILITY` in redactLakeForActor.ts and for the same reason: a list of
@@ -2289,6 +2571,7 @@ const LAKE_CONFIG_FIELD_AUDIT = {
2289
2571
  organizationId: "audited",
2290
2572
  isPublic: "audited",
2291
2573
  auditQueryTextEnabled: "audited",
2574
+ lakeMemoryEnabled: "audited",
2292
2575
  status: "audited",
2293
2576
  createdByUserId: "audited",
2294
2577
  lastUpdatedByUserId: "excluded",
@@ -2300,7 +2583,10 @@ const LAKE_CONFIG_FIELD_AUDIT = {
2300
2583
  filesDeletedAt: "excluded",
2301
2584
  filesArchivedAt: "excluded",
2302
2585
  lakeMemoryExtractionAt: "excluded",
2303
- lakeMemoryCursor: "excluded"
2586
+ lakeMemoryCursor: "excluded",
2587
+ lakeMemoryPurgedAt: "excluded",
2588
+ inconsistencyReport: "excluded",
2589
+ inconsistencyComputedAt: "excluded"
2304
2590
  };
2305
2591
  [...Object.keys(LAKE_CONFIG_FIELD_AUDIT).filter((field) => LAKE_CONFIG_FIELD_AUDIT[field] === "audited")];
2306
2592
  /**
@@ -2862,6 +3148,7 @@ function toModelInfo(record) {
2862
3148
  disabled,
2863
3149
  disabledReason: record.disabledReason ?? record.autoDisabledReason ?? (retired ? "retired by the provider" : void 0),
2864
3150
  deprecationDate: record.lifecycle?.deprecationDate,
3151
+ replacedBy: record.lifecycle?.replacedBy,
2865
3152
  trainingCutoff: record.trainingCutoff,
2866
3153
  releaseDate: record.releaseDate,
2867
3154
  logoFile: record.logoFile,
@@ -2894,6 +3181,7 @@ const VENDOR_BY_BACKEND = {
2894
3181
  ["gemini"]: "google",
2895
3182
  ["xai"]: "xai",
2896
3183
  ["kimi"]: "moonshotai",
3184
+ ["deepseek"]: "deepseek",
2897
3185
  ["bfl"]: "black-forest-labs",
2898
3186
  ["aws"]: "amazon",
2899
3187
  ["voyageai"]: "voyageai",
@@ -2962,10 +3250,13 @@ function toModelRecord(info) {
2962
3250
  supportsTools: info.supportsTools,
2963
3251
  supportsImageVariation: info.supportsImageVariation,
2964
3252
  supportsSafetyTolerance: info.supportsSafetyTolerance,
2965
- lifecycle: info.deprecationDate ? {
2966
- status: "deprecated",
2967
- deprecationDate: info.deprecationDate
2968
- } : { status: "active" },
3253
+ lifecycle: {
3254
+ ...info.deprecationDate ? {
3255
+ status: "deprecated",
3256
+ deprecationDate: info.deprecationDate
3257
+ } : { status: "active" },
3258
+ ...info.replacedBy ? { replacedBy: info.replacedBy } : {}
3259
+ },
2969
3260
  description: info.description,
2970
3261
  logoFile: info.logoFile,
2971
3262
  rank: info.rank,
@@ -3094,6 +3385,71 @@ const usdToCreditsStochastic = (usd, rng = cryptoUniform, rate = CREDITS_PER_USD
3094
3385
  const fraction = raw - base;
3095
3386
  return base + (rng() < fraction ? 1 : 0);
3096
3387
  };
3388
+ /**
3389
+ * Output tokens the pre-flight credit hold prices when a request's max_tokens
3390
+ * ceiling exceeds it.
3391
+ *
3392
+ * max_tokens is a *ceiling*, not a prediction: adaptive reasoning models stop at
3393
+ * end_turn well short of it, so pricing the hold at the full window (128K on the
3394
+ * flagship models) reserved several thousand credits per turn regardless of answer
3395
+ * length. The excess was always refunded at settlement, but the hold IS the
3396
+ * insufficient-funds gate, so users whose balance sat between their real cost and
3397
+ * the worst case were falsely blocked.
3398
+ *
3399
+ * 16K covers the realistic long answer with headroom - the largest replies seen in
3400
+ * practice are HTML-artifact turns at roughly 10-11K output tokens (see
3401
+ * buildThinkingParams in llm-adapters/thinkingParams.ts, whose max_tokens floor was
3402
+ * sized off the same measurement).
3403
+ *
3404
+ * UNDER-RESERVATION IS ACCEPTED, NOT PREVENTED. A turn that emits more than this
3405
+ * settles as a shortfall debit at reconciliation, which is already a supported path
3406
+ * (provider-basis settlement could always exceed a hold priced on the local
3407
+ * estimate). The two sites clamp the shortfall differently: chat's
3408
+ * computeSettlementDelta (services/llm/ChatCompletionProcess.ts) floors the debit at
3409
+ * the balance snapshot taken at its OWN admission and reports the remainder as
3410
+ * writtenOffCredits; cliCompletions.ts does an unclamped $inc on success and only
3411
+ * logs an ALERT if it lands negative. Neither is a non-negativity guarantee: the
3412
+ * chat snapshot predates any sibling turn's spend (see that function's own "best
3413
+ * effort" note), and when the shortfall still fits the stale snapshot the debit
3414
+ * applies in full - writtenOffCredits 0, and no BILLING_SHORTFALL_CLAMP log - so a
3415
+ * concurrent holder can land negative there too. Either way the resulting balance
3416
+ * fails the *next* turn's own admission gate - but that bound is per turn, not per
3417
+ * holder: turns admitted concurrently are each checked against the balance at their
3418
+ * own admission and settle against a snapshot that predates their siblings' spend,
3419
+ * so a holder running turns in parallel can be shorted once per in-flight turn, not
3420
+ * once total.
3421
+ */
3422
+ const PREFLIGHT_RESERVATION_OUTPUT_TOKENS = 16384;
3423
+ /**
3424
+ * Reservation ceiling for models that spend reasoning tokens inside their output
3425
+ * budget (see reasonsWithinOutputBudget in llm-adapters/thinkingParams.ts). Those
3426
+ * tokens bill as output on top of the visible answer, so the 16K figure above -
3427
+ * which was measured on visible artifact size alone - under-reserves them badly.
3428
+ *
3429
+ * Deliberately below ADAPTIVE_THINKING_MAX_TOKENS_FLOOR (64K), which sizes the
3430
+ * request's real ceiling and therefore has to cover the worst case: exceeding it
3431
+ * truncates a reply mid-tag, an unrecoverable failure. A hold has no such duty -
3432
+ * exceeding it settles as a shortfall debit - so it is sized for the long turn
3433
+ * (roughly 3x the largest observed visible answer, leaving the rest for the trace)
3434
+ * rather than the worst one, which is what keeps the gate off affordable requests.
3435
+ */
3436
+ const PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS = 32768;
3437
+ /**
3438
+ * Output-token figure to price a pre-flight credit hold at, given the max_tokens
3439
+ * this request will actually send. Never raises the caller's ceiling: a request
3440
+ * that asks for less than the cap holds only what it can possibly spend.
3441
+ *
3442
+ * Reserving only, never gating: the per-member org credit cap is still priced on
3443
+ * the unshrunk ceiling at both call sites, since that check has no settlement
3444
+ * counterpart to correct an under-estimate. That is a strictly larger figure than
3445
+ * the hold, not a true upper bound on the turn - it prices one model round trip,
3446
+ * at the uncached input rate, on the primary model - so it still under-counts a
3447
+ * multi-round tool loop, a cache-write turn, or a fallback hop onto pricier pricing.
3448
+ *
3449
+ * @param reasonsWithinOutputBudget - reasonsWithinOutputBudget(modelInfo); passed as
3450
+ * a boolean because common cannot import llm-adapters.
3451
+ */
3452
+ const reservationOutputTokens = (requestedMaxTokens, reasonsWithinOutputBudget = false) => Math.min(requestedMaxTokens, reasonsWithinOutputBudget ? PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS : PREFLIGHT_RESERVATION_OUTPUT_TOKENS);
3097
3453
  z.enum([
3098
3454
  "openai",
3099
3455
  "test",
@@ -3101,6 +3457,58 @@ z.enum([
3101
3457
  "xai",
3102
3458
  "gemini"
3103
3459
  ]);
3460
+ z$1.enum([
3461
+ "inject",
3462
+ "auto-fire",
3463
+ "hidden"
3464
+ ]);
3465
+ /**
3466
+ * Modes acceptable at AUTHORING time. 'hidden' is intentionally excluded until
3467
+ * the host has true hidden-send support - accepting it would persist a value
3468
+ * that silently behaves as 'auto-fire' (a surprising downgrade). It stays in
3469
+ * ExecutionModeSchema/the stored enum for forward-compat.
3470
+ */
3471
+ const AuthorableExecutionModeSchema = z$1.enum(["inject", "auto-fire"]);
3472
+ /**
3473
+ * Tools a prompt may require - constrained to the host's closed tool set, MINUS
3474
+ * integration-gated tools that act on the caller's own credentials/account. A
3475
+ * shared system prompt must not be able to inject e.g. blog-publishing into a
3476
+ * non-author's session via requiredTools. (Per-user entitlement of the remaining
3477
+ * tools is still the chat pipeline's responsibility - see follow-up note in the
3478
+ * briefcase blueprint; this allowlist is the storage-layer floor.)
3479
+ */
3480
+ const BRIEFCASE_DISALLOWED_TOOLS = [
3481
+ "blog_publish",
3482
+ "blog_edit",
3483
+ "blog_draft"
3484
+ ];
3485
+ const BriefcaseRequiredToolsSchema = z$1.array(b4mLLMTools.refine((t) => !BRIEFCASE_DISALLOWED_TOOLS.includes(t), "This tool is not permitted in a briefcase prompt")).max(16);
3486
+ z$1.string().regex(/^[a-f0-9]{24}$/i, "Invalid prompt id");
3487
+ /** Shared cap for every free-text prompt body a caller can supply. Exported so the
3488
+ * caller-prompt caps on the chat request, the invoke params and the CLI tool import it
3489
+ * rather than each restating the literal. */
3490
+ const PROMPT_TEXT_MAX = 16e3;
3491
+ const TAGS_MAX = 20;
3492
+ z$1.object({
3493
+ type: z$1.string().min(1).max(100),
3494
+ name: z$1.string().min(1).max(200),
3495
+ description: z$1.string().max(500).optional(),
3496
+ promptText: z$1.string().min(1).max(PROMPT_TEXT_MAX),
3497
+ tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
3498
+ executionMode: AuthorableExecutionModeSchema.optional(),
3499
+ requiredTools: BriefcaseRequiredToolsSchema.optional()
3500
+ }).partial();
3501
+ /**
3502
+ * One catalog sub-query. Exactly one selector is used, in precedence order:
3503
+ * `personal` (resolved to the caller server-side) > `tags` > `type`.
3504
+ */
3505
+ const PromptBatchQuerySchema = z$1.object({
3506
+ key: z$1.string().min(1).max(100),
3507
+ tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
3508
+ type: z$1.string().max(100).optional(),
3509
+ personal: z$1.boolean().optional()
3510
+ });
3511
+ z$1.object({ queries: z$1.array(PromptBatchQuerySchema).min(1).max(32).refine((qs) => new Set(qs.map((q) => q.key)).size === qs.length, { message: "Batch query keys must be unique" }) });
3104
3512
  z$1.object({
3105
3513
  sessionId: z$1.string().nullish(),
3106
3514
  message: z$1.string(),
@@ -3116,7 +3524,7 @@ z$1.object({
3116
3524
  wait: z$1.boolean().prefault(false),
3117
3525
  enableTools: z$1.boolean().prefault(false),
3118
3526
  toolMode: z$1.enum(["fast", "smart"]).optional(),
3119
- tools: z$1.array(z$1.string()).optional(),
3527
+ tools: z$1.array(z$1.string()).optional().describe("Explicit tool ids to offer the model. A non-empty array enables tools on its own; no companion `toolMode` or `enableTools` is required. Merged with the auto-selected set under `toolMode: \"smart\"`, and ignored under `toolMode: \"fast\"`. This list ADDS to what is offered rather than restricting it - the server still offers tools of its own (for example knowledge retrieval when the session has reachable documents). Unrecognized ids are dropped rather than rejecting the request; the response reports the surviving set as `tools.effectiveTools` and the ids that are not tools in this deployment as `tools.unrecognizedTools`, since no endpoint enumerates the valid ids. `unrecognizedTools` is reported under every `toolMode` - it describes the ids, not what the mode did with them - so under `toolMode: \"fast\"` an id can be absent from `effectiveTools` (the mode discarded it) without being unrecognized. Both reported lists are deduplicated, and `unrecognizedTools` names at most the first 10 distinct ids."),
3120
3528
  enableQuestMaster: z$1.boolean().optional(),
3121
3529
  enableMementos: z$1.boolean().optional(),
3122
3530
  enableAgents: z$1.boolean().optional(),
@@ -3125,8 +3533,10 @@ z$1.object({
3125
3533
  "grounded",
3126
3534
  "surface"
3127
3535
  ]).optional(),
3536
+ skip_auto_offers: z$1.boolean().optional().describe("Suppress tools the server would otherwise attach on its own for this session (the knowledge-base search offer, in-app view navigation, blog drafting/editing/publishing, and skill invocation). Tools you request explicitly are unaffected. One system-prompt block goes with them: withholding in-app view navigation also drops the view-registry block that exists only to describe it. No other prompt content changes. This does not switch off retrieval: a session with forced knowledge retrieval still retrieves, and documents already attached to the session are still placed in the prompt directly. Any promptMode suppresses these too, so false has no effect alongside one."),
3128
3537
  includePromptDetails: z$1.boolean().optional(),
3129
- includeSystemPrompt: z$1.boolean().optional()
3538
+ includeSystemPrompt: z$1.boolean().optional(),
3539
+ systemPrompt: z$1.string().max(PROMPT_TEXT_MAX).optional().describe("System-prompt text for this request only, never persisted. Rendered as a defended block appended after every other system-prompt source, with prose instructing the model to defer to organization, session and data-lake guidance. Over the cap is a 422, never truncated.")
3130
3540
  });
3131
3541
  z$1.object({
3132
3542
  id: z$1.string(),
@@ -3135,6 +3545,12 @@ z$1.object({
3135
3545
  timestamp: z$1.string(),
3136
3546
  model: z$1.string(),
3137
3547
  message: z$1.string().optional(),
3548
+ tools: z$1.object({
3549
+ toolMode: z$1.enum(["fast", "smart"]).optional(),
3550
+ autoSelectedTools: z$1.array(z$1.string()).optional(),
3551
+ effectiveTools: z$1.array(z$1.string()),
3552
+ unrecognizedTools: z$1.array(z$1.string()).optional()
3553
+ }).optional(),
3138
3554
  tracking_info: z$1.object({
3139
3555
  quest_id: z$1.string(),
3140
3556
  check_status_url: z$1.string(),
@@ -3357,13 +3773,93 @@ const AGENT_EXECUTION_STATUSES = [
3357
3773
  "failed",
3358
3774
  "aborted"
3359
3775
  ];
3776
+ /**
3777
+ * Operator allow-list for the client-authored Mongo filter carried on a `subscribe_query` frame.
3778
+ *
3779
+ * The WS data-subscribe handler forwards that filter to `Model.find` and persists it on the
3780
+ * QuerySubscription the separate subscriber-fanout service replays, so an unconstrained
3781
+ * `$`-prefixed key lets any authenticated socket ask the database to run server-side JavaScript
3782
+ * (`$where`, `$expr` + `$function`) or an unbounded `$regex` - either of which pins a pooled
3783
+ * connection for as long as it runs.
3784
+ *
3785
+ * Live subscriptions only ever need equality, membership and range matching, so anything outside
3786
+ * this list is REFUSED rather than stripped: silently dropping an operator would quietly widen the
3787
+ * document set the caller ends up subscribed to.
3788
+ *
3789
+ * The scan is deliberately structural rather than a model of Mongo's grammar - it flags any
3790
+ * `$`-prefixed key anywhere in the tree that isn't allow-listed. It can therefore accept an
3791
+ * allow-listed operator in a position Mongo would treat as a literal sub-document field name, but
3792
+ * that is inert; what matters is that no disallowed operator can reach the driver.
3793
+ */
3794
+ /** Combinators whose operands are themselves filters. */
3795
+ const SUBSCRIPTION_FILTER_LOGICAL_OPERATORS = [
3796
+ "$and",
3797
+ "$or",
3798
+ "$nor",
3799
+ "$not"
3800
+ ];
3801
+ /** Value-level operators a subscription filter may use. */
3802
+ const SUBSCRIPTION_FILTER_VALUE_OPERATORS = [
3803
+ "$eq",
3804
+ "$ne",
3805
+ "$gt",
3806
+ "$gte",
3807
+ "$lt",
3808
+ "$lte",
3809
+ "$in",
3810
+ "$nin",
3811
+ "$exists",
3812
+ "$type",
3813
+ "$size",
3814
+ "$all",
3815
+ "$elemMatch"
3816
+ ];
3817
+ const ALLOWED_OPERATORS = /* @__PURE__ */ new Set([...SUBSCRIPTION_FILTER_LOGICAL_OPERATORS, ...SUBSCRIPTION_FILTER_VALUE_OPERATORS]);
3818
+ /**
3819
+ * Returns a dotted path for every part of `filter` a subscription may not send: a disallowed
3820
+ * `$`-prefixed key, a `RegExp` operand, or nesting past {@link SUBSCRIPTION_FILTER_MAX_DEPTH}.
3821
+ * An empty array means the filter is safe to forward to the database.
3822
+ */
3823
+ function findDisallowedSubscriptionFilterKeys(filter) {
3824
+ const violations = [];
3825
+ const walk = (node, path, depth) => {
3826
+ if (depth > 12) {
3827
+ violations.push(`${path} (nested deeper than 12)`);
3828
+ return;
3829
+ }
3830
+ if (node instanceof RegExp) {
3831
+ violations.push(`${path} (regular expression)`);
3832
+ return;
3833
+ }
3834
+ if (Array.isArray(node)) {
3835
+ node.forEach((entry, i) => walk(entry, `${path}[${i}]`, depth + 1));
3836
+ return;
3837
+ }
3838
+ if (node === null || typeof node !== "object") return;
3839
+ for (const [key, value] of Object.entries(node)) {
3840
+ const childPath = path ? `${path}.${key}` : key;
3841
+ if (key.startsWith("$") && !ALLOWED_OPERATORS.has(key)) {
3842
+ violations.push(childPath);
3843
+ continue;
3844
+ }
3845
+ walk(value, childPath, depth + 1);
3846
+ }
3847
+ };
3848
+ walk(filter, "", 0);
3849
+ return violations;
3850
+ }
3360
3851
  const DataSubscribeRequestAction = z$1.object({
3361
3852
  action: z$1.literal("subscribe_query"),
3362
3853
  accessToken: z$1.string().optional(),
3363
3854
  subscriptionId: z$1.string(),
3364
3855
  collectionName: z$1.string(),
3365
- query: z$1.looseObject({}),
3366
- fields: z$1.looseObject({}),
3856
+ query: z$1.looseObject({}).superRefine((filter, ctx) => {
3857
+ for (const key of findDisallowedSubscriptionFilterKeys(filter)) ctx.addIssue({
3858
+ code: "custom",
3859
+ message: `Disallowed subscription filter: ${key}`
3860
+ });
3861
+ }),
3862
+ fields: z$1.record(z$1.string(), z$1.union([z$1.boolean(), z$1.number()])),
3367
3863
  fetchInitialData: z$1.boolean().prefault(true).optional(),
3368
3864
  clientId: z$1.string().optional()
3369
3865
  });
@@ -3450,6 +3946,17 @@ const DataSubscriptionUpdateAction = z$1.object({
3450
3946
  id: z$1.string()
3451
3947
  })
3452
3948
  });
3949
+ /**
3950
+ * Server -> Client: a `subscribe_query` frame was refused (the operator allow-list, `fields`
3951
+ * validation) or its initial fetch was aborted (`maxTimeMS`). Neither failure throws an
3952
+ * UnauthorizedError/JsonWebTokenError, so withWebSocketContext's status code never reaches the
3953
+ * client as a frame - this is the only signal the caller gets that the subscription never took.
3954
+ */
3955
+ const DataSubscribeErrorAction = z$1.object({
3956
+ action: z$1.literal("data_subscribe_error"),
3957
+ subscriptionId: z$1.string(),
3958
+ error: z$1.string()
3959
+ });
3453
3960
  const LLMStatusUpdateAction = z$1.object({
3454
3961
  action: z$1.literal("llm_status_update"),
3455
3962
  status: z$1.string().nullable(),
@@ -3532,6 +4039,7 @@ const QuestExportProgressAction = z$1.object({
3532
4039
  downloadUrl: z$1.string().optional(),
3533
4040
  filename: z$1.string().optional(),
3534
4041
  errorMessage: z$1.string().optional(),
4042
+ droppedQuestCount: z$1.number().optional(),
3535
4043
  clientId: z$1.string().optional()
3536
4044
  });
3537
4045
  const SpiderProgressUpdateAction = z$1.object({
@@ -4682,6 +5190,7 @@ const OptiHashiRunUpdatedAction = z$1.object({
4682
5190
  });
4683
5191
  z$1.discriminatedUnion("action", [
4684
5192
  DataSubscriptionUpdateAction,
5193
+ DataSubscribeErrorAction,
4685
5194
  InboxRefetchAction,
4686
5195
  LLMStatusUpdateAction,
4687
5196
  InvitesRefetchAction,
@@ -4739,6 +5248,126 @@ z$1.discriminatedUnion("action", [
4739
5248
  PermissionRequestAction,
4740
5249
  ReconnectResultAction
4741
5250
  ]);
5251
+ z$1.object({
5252
+ session_id: z$1.string().min(1),
5253
+ message: z$1.string().min(1),
5254
+ /** Falls back to the deployment's default chat model when omitted. */
5255
+ model: z$1.string().optional(),
5256
+ /**
5257
+ * Run as a specific persisted agent. Omit to let the executor pick the profile for
5258
+ * the session: a session on a dedicated surface gets that surface's own profile,
5259
+ * otherwise a synthetic one built from admin orchestration defaults. Omitting this
5260
+ * is what reproduces the product UI's Agent Mode toggle.
5261
+ */
5262
+ agent_id: z$1.string().optional(),
5263
+ /**
5264
+ * Bill this run to an organization's credit pool. The caller must belong to it; a
5265
+ * non-member gets 404. Omit to bill the caller personally.
5266
+ */
5267
+ organization_id: z$1.string().optional(),
5268
+ /**
5269
+ * Tool-id allowlist for the run. Omit to use the resolved profile's own list.
5270
+ *
5271
+ * Doubles as pre-approval: REST runs have no interactive client to answer a
5272
+ * permission prompt, so tools named here are treated as approved. A run that calls
5273
+ * an approval-gated tool NOT named here fails with that tool named in `error`.
5274
+ */
5275
+ tools: z$1.array(z$1.string()).optional(),
5276
+ /**
5277
+ * Hard ceiling on ReAct iterations. Each one is a full LLM round-trip, so the cap is
5278
+ * bounded at 100 regardless of what the profile would allow.
5279
+ */
5280
+ max_iterations: z$1.number().int().positive().max(100).optional(),
5281
+ temperature: z$1.number().min(0).max(2).optional(),
5282
+ max_tokens: z$1.number().int().positive().optional(),
5283
+ thinking: z$1.object({
5284
+ enabled: z$1.boolean(),
5285
+ /** Bounded at 32000: Anthropic rejects rather than clamps an oversized budget. */
5286
+ budget_tokens: z$1.number().int().positive().max(32e3).optional()
5287
+ }).optional(),
5288
+ /** Per-message file attachments (fabFile ids), materialized into the first iteration. */
5289
+ file_ids: z$1.array(z$1.string()).optional(),
5290
+ /** Workbench-level file ids for the session, forwarded as a dispatch-time snapshot. */
5291
+ session_file_ids: z$1.array(z$1.string()).optional(),
5292
+ enable_mementos: z$1.boolean().optional(),
5293
+ enable_lattice: z$1.boolean().optional(),
5294
+ /**
5295
+ * Opt out of the artifact-emission prompt and artifact persistence for this run.
5296
+ *
5297
+ * ANDed with the deployment's admin `EnableArtifacts` setting, so this can only ever
5298
+ * withhold artifacts, never force them on. Omitting it means "no preference" and
5299
+ * leaves the admin setting as the only gate; only an explicit `false` opts out.
5300
+ * Inherited by any subagent this run dispatches, so a delegating agent cannot route
5301
+ * around the opt-out.
5302
+ *
5303
+ * Worth setting on a REST run: the emission prompt costs roughly 2.8k tokens per
5304
+ * iteration, and nothing on this transport renders an artifact back to a human.
5305
+ */
5306
+ enable_artifacts: z$1.boolean().optional()
5307
+ });
5308
+ z$1.object({
5309
+ id: z$1.string(),
5310
+ status: z$1.literal("pending"),
5311
+ session_id: z$1.string(),
5312
+ model: z$1.string(),
5313
+ timestamp: z$1.string(),
5314
+ tracking_info: z$1.object({
5315
+ execution_id: z$1.string(),
5316
+ /**
5317
+ * The chat-history Quest holding the prompt, which gains the reply when the run
5318
+ * completes. Absent when that best-effort write failed; the run still proceeds.
5319
+ */
5320
+ quest_id: z$1.string().optional(),
5321
+ poll_url: z$1.string()
5322
+ })
5323
+ });
5324
+ /** One step of a published reasoning trace - a public projection of `IAgentStep`. */
5325
+ const AgentExecutionStepSchema = z$1.object({
5326
+ type: z$1.enum([
5327
+ "thought",
5328
+ "action",
5329
+ "observation",
5330
+ "final_answer"
5331
+ ]),
5332
+ content: z$1.string(),
5333
+ /** 0-indexed iteration. Absent on traces checkpointed before the field existed. */
5334
+ iteration: z$1.number().int().nonnegative().optional(),
5335
+ /** Set on `action` steps: the tool the agent invoked. */
5336
+ tool_name: z$1.string().optional()
5337
+ });
5338
+ z$1.object({
5339
+ id: z$1.string(),
5340
+ status: z$1.enum([
5341
+ "pending",
5342
+ "running",
5343
+ "continuing",
5344
+ "awaiting_permission",
5345
+ "awaiting_subagent",
5346
+ "awaiting_dag_children",
5347
+ "paused",
5348
+ "completed",
5349
+ "failed",
5350
+ "aborted"
5351
+ ]),
5352
+ session_id: z$1.string().nullable(),
5353
+ answer: z$1.string().nullable(),
5354
+ /**
5355
+ * Why the run ended without an answer. Set only on `failed`; null otherwise.
5356
+ * Without this a caller polling a terminal run sees `failed` + a null answer and
5357
+ * cannot tell an approval-gated tool from a model error from a timeout.
5358
+ *
5359
+ * Approval-gate failures name the offending tool, since that is what the caller acts
5360
+ * on. Everything else is reduced to a coarse category (billing, rate limit, timeout,
5361
+ * auth) or a generic message: the stored reason is a raw internal exception, and
5362
+ * those carry infrastructure identifiers. Full detail stays in the server logs.
5363
+ */
5364
+ error: z$1.string().nullable(),
5365
+ steps: z$1.array(AgentExecutionStepSchema),
5366
+ total_iterations: z$1.number().nullable(),
5367
+ created_at: z$1.string(),
5368
+ updated_at: z$1.string()
5369
+ });
5370
+ z$1.object({ id: z$1.string().min(1) });
4742
5371
  /**
4743
5372
  * Tool schema matching ICompletionOptionTools.toolSchema. The Zod surface only
4744
5373
  * covers wire-format fields (toolFn is server-side). Replaces the historical
@@ -5813,43 +6442,129 @@ z$2.object({
5813
6442
  */
5814
6443
  const OVERSIZED_PASSAGE_TOKEN_THRESHOLD = 1500;
5815
6444
  /**
5816
- * `FabFile.notes` marker written when the data-lake convergence kill switch abandons a vectorize
5817
- * (#1676). The file keeps its chunks but has no vectors, so it is unsearchable until re-indexed, and
5818
- * it does NOT auto-resume.
6445
+ * Why a file's chunk/vector pipeline is STALLED, stored in `FabFile.chunkStallReason`.
6446
+ *
6447
+ * - `vectorizePaused`: the data-lake convergence kill switch abandoned a vectorize (#1676). The
6448
+ * file keeps its chunks but has no vectors, so it is unsearchable until re-indexed.
6449
+ * - `rechunkPaused`: the OTHER half of the same switch dropped a re-chunk before it ran
6450
+ * (#1676/#1681). The damage is worse - the producer resets a wave's chunk state BEFORE the
6451
+ * messages are handled, so a file halted here has NO chunks at all.
6452
+ * - `unchunkedPaused`: the same chunk half, on a file that never had passages to lose. The rescue
6453
+ * sweep selects on `chunkCount: 0` and enqueues without resetting anything, so a file it routes
6454
+ * into the halt branch arrives already empty. The halted STATE is identical to `rechunkPaused`
6455
+ * and every reader keys on both (CHUNKLESS_STALL_REASONS) - the split exists so the owner is not
6456
+ * told that passages were removed which never existed.
6457
+ *
6458
+ * None auto-resumes; each needs a reprocess or a lifted switch.
5819
6459
  *
5820
- * Lives here rather than beside its writer (apps/client fabFileVectorize) because it is a
5821
- * cross-layer contract: the queue handler writes it and the lake-health evaluator
5822
- * (constants/lakeHealth.ts) reads it to tell a permanently-stalled file from one still in flight.
5823
- * b4m-core cannot import from apps/client, so a copy there would have to drift silently.
6460
+ * Without a marker the state is misread by every surface at once, which is the failure the field
6461
+ * exists to prevent: `chunkCount: 0` with `error: null` reads as an image or a pending upload, so
6462
+ * health drops it from the denominator, convergence grades it `conformant` (its stale stamp still
6463
+ * matches), and search does not withhold it because it is not "in flight". The file's passages are
6464
+ * simply gone and nothing REPORTS it. The rescue sweep is the one exception and deliberately so: an
6465
+ * unmarked file matches its filter and gets re-chunked, which is repair rather than reporting. That
6466
+ * is why the sweep excludes a stalled file only while the switch is ON - see
6467
+ * buildFabFileChunkScanFilter.
6468
+ *
6469
+ * A dedicated field rather than prose in `FabFile.notes` (#2016): `notes` is the USER's note, and
6470
+ * while the markers lived there every writer of the field clobbered the others - a "Rebuild
6471
+ * passages" wave silently deleted whatever the owner had typed.
6472
+ *
6473
+ * Lives here rather than beside its writers (apps/client's chunk and vectorize handlers) because it
6474
+ * is a cross-layer contract: the queue handlers write it and b4m-core's evaluators
6475
+ * (constants/lakeHealth.ts, constants/lakeConvergence.ts, dataLakeService/retrievalUnavailable.ts)
6476
+ * read it to tell a permanently-stalled file from one still in flight. b4m-core cannot import from
6477
+ * apps/client, so a copy there would have to drift silently.
6478
+ */
6479
+ const CHUNK_STALL_REASONS = [
6480
+ "vectorizePaused",
6481
+ "rechunkPaused",
6482
+ "unchunkedPaused"
6483
+ ];
6484
+ /**
6485
+ * Whether a file is stalled by the convergence kill switch, by any arm. THE predicate every
6486
+ * reader uses, so adding a stall reason reaches health, convergence and retrieval without separate
6487
+ * comparisons drifting apart. Also the in-memory mirror of a Mongo
6488
+ * `chunkStallReason: { $in: [...CHUNK_STALL_REASONS] }`.
5824
6489
  */
5825
- const CONVERGENCE_PAUSED_NOTE = "Indexing paused by the data-lake convergence kill switch - reprocess to complete.";
6490
+ function isChunkStalled(reason) {
6491
+ return CHUNK_STALL_REASONS.includes(reason);
6492
+ }
5826
6493
  /**
5827
- * `FabFile.notes` marker for the OTHER half of the same kill switch: a re-chunk dropped before it
5828
- * ran (#1676/#1681). Distinct from `CONVERGENCE_PAUSED_NOTE` because the damage is worse and the
5829
- * wording has to say so - the producer resets a wave's chunk state BEFORE the messages are handled,
5830
- * so a file halted here has NO chunks at all rather than chunks without vectors.
6494
+ * Which reasons leave the file with NO passages, as opposed to passages with no vectors. A `Record`
6495
+ * over every reason rather than a hand-written subset array: a new stall reason then cannot compile
6496
+ * until it is classified, where a member missing from a literal array would just make a health count
6497
+ * silently wrong.
6498
+ */
6499
+ const STALL_LEAVES_NO_PASSAGES = {
6500
+ vectorizePaused: false,
6501
+ rechunkPaused: true,
6502
+ unchunkedPaused: true
6503
+ };
6504
+ CHUNK_STALL_REASONS.filter((reason) => STALL_LEAVES_NO_PASSAGES[reason]);
6505
+ /**
6506
+ * Owner-facing prose for a stall reason, and the ONLY place it is worded.
5831
6507
  *
5832
- * Without a marker this state is invisible to every surface at once, which is the failure it exists
5833
- * to prevent: `chunkCount: 0` with `error: null` reads as an image or a pending upload, so health
5834
- * drops it from the denominator, convergence grades it `conformant` (its stale stamp still matches),
5835
- * search does not withhold it because it is not "in flight", and the rescue sweep's own filter
5836
- * passes over it. The file's passages are simply gone and nothing reports it.
6508
+ * `vectorizePaused` and `rechunkPaused` are the exact strings those markers used while they lived in
6509
+ * `notes`, which is also what the #2016 migration matches on to derive the field for existing rows -
6510
+ * do not reword either without updating it. `unchunkedPaused` postdates that migration and was never
6511
+ * written to `notes`, so its wording is free to change: nothing matches on it.
6512
+ */
6513
+ const CHUNK_STALL_NOTICES = {
6514
+ vectorizePaused: "Indexing paused by the data-lake convergence kill switch - reprocess to complete.",
6515
+ rechunkPaused: "Re-chunking paused by the data-lake convergence kill switch - its passages were removed and are rebuilt when convergence resumes.",
6516
+ unchunkedPaused: "Chunking paused by the data-lake convergence kill switch - this file has no passages yet and they are built when convergence resumes."
6517
+ };
6518
+ CHUNK_STALL_NOTICES.vectorizePaused;
6519
+ CHUNK_STALL_NOTICES.rechunkPaused;
6520
+ /**
6521
+ * TRANSITIONAL, and the ONE stall predicate every RETRIEVAL path must use until #2016's migration
6522
+ * has run in every environment. Reads the new field, then falls back to the legacy prose that the
6523
+ * pre-migration rows still carry in `notes`.
6524
+ *
6525
+ * It exists for the FORWARD window only: `migratorInvocation` is a `dependsOn` of the web stack
6526
+ * only (infra/web.ts); the queue stack has none, so the executor can serve forced retrieval and
6527
+ * `knowledge_base_search` while rows still carry the marker in `notes` and no `chunkStallReason`. A
6528
+ * row stalled by the chunk arm then reads as a plain unindexed file: `isRetrievalExcluded` drops it
6529
+ * upstream of the withhold on a vectorizedOnly lake, and `partitionByIndexAvailability` calls it
6530
+ * servable everywhere else. The turn answers around a passage-less file and reports FULL coverage -
6531
+ * the silent degradation this whole path exists to prevent.
6532
+ *
6533
+ * A code ROLLBACK is the mirror image and this arm CANNOT cover it: the rows are already migrated
6534
+ * (`chunkStallReason` set, `notes` unset) and the code restored is pre-#2016, which does not contain
6535
+ * this function. Nothing reverts the data on its own either - `migratorInvocation` only ever runs
6536
+ * `up` and `migrate down` is a manual CLI step - so `migrate down` is a REQUIRED step of any
6537
+ * rollback past #2016, not an optional tidy-up. What this arm does buy is that `down()` is safe to
6538
+ * run FIRST: whichever stack is still new keeps honoring the prose it restores, so a staggered
6539
+ * rollback has no window where a restored marker is invisible. `down()` is a PARTIAL restore
6540
+ * though - it skips a row whose owner typed a note after `up()`, and that row grades as unstalled
6541
+ * on both stacks once the field is dropped. See its own comment.
6542
+ *
6543
+ * Deliberately NOT used by the grading/health/UI readers: they are gated behind the web stack, and
6544
+ * a legacy row there renders the notice line AND the identical text as the owner's note.
5837
6545
  *
5838
- * Same cross-layer reason as the constant above for living here: the queue handler writes it and
5839
- * b4m-core's evaluators read it, and b4m-core cannot import from apps/client.
6546
+ * Mirrored in Mongo by `buildFabFileSearchQuery`'s `vectorizedOnly` exemption. Delete the legacy arm
6547
+ * from both together, one release after the migration has landed everywhere.
6548
+ *
6549
+ * Pinned to the two reasons the migration backfilled rather than every notice: `unchunkedPaused`
6550
+ * postdates it, so no row carries its prose, and including it would read an owner who happens to type
6551
+ * that sentence into `notes` as stalled.
5840
6552
  */
5841
- const CONVERGENCE_PAUSED_CHUNK_NOTE = "Re-chunking paused by the data-lake convergence kill switch - its passages were removed and are rebuilt when convergence resumes.";
6553
+ const LEGACY_CHUNK_STALL_NOTES = [CHUNK_STALL_NOTICES.vectorizePaused, CHUNK_STALL_NOTICES.rechunkPaused];
6554
+ function isChunkStalledFile(file) {
6555
+ return isChunkStalled(file.chunkStallReason) || LEGACY_CHUNK_STALL_NOTES.includes(file.notes ?? "");
6556
+ }
5842
6557
  /**
5843
6558
  * `FabFile.chunkRebuildRequestedAt`: stamped by `resetChunkStateByIds` in the SAME write that
5844
6559
  * clears a file's chunk rollups, so "this file's passages are being rebuilt" can never be lost the
5845
6560
  * way the pair of steps that creates the state can be. The reset and the queue send are two
5846
6561
  * operations - kill the producer between them, or lose the consumer's marker write, and the file
5847
- * sits at `chunkCount: 0` with `error: null` and `notes: ''`, a shape indistinguishable from an
6562
+ * sits at `chunkCount: 0` with `error: null` and no stall reason, a shape indistinguishable from an
5848
6563
  * image or a still-uploading row. It then drops out of lake health's denominator, out of the
5849
6564
  * convergence plan and out of the retrieval withhold at the same moment: every rollup says its
5850
6565
  * passages are gone, and nothing reports it.
5851
6566
  *
5852
- * Deliberately NOT `CONVERGENCE_PAUSED_CHUNK_NOTE` pre-written by the producer, which is the obvious
6567
+ * Deliberately NOT the `rechunkPaused` stall reason pre-written by the producer, which is the obvious
5853
6568
  * fix and the wrong one: that marker means "halted, needs an administrator", so a file awaiting an
5854
6569
  * ORDINARY rebuild would read to every reader as permanently paused for the whole rebuild - search
5855
6570
  * would tell readers it does not return on its own, health would hard-fail P3, and "Rebuild
@@ -5861,9 +6576,9 @@ const CONVERGENCE_PAUSED_CHUNK_NOTE = "Re-chunking paused by the data-lake conve
5861
6576
  * upgrade therefore degrades to mislabelled-but-visible rather than invisible, which is the trade
5862
6577
  * this field exists to make - invisibility is the real harm, labelling is secondary.
5863
6578
  *
5864
- * A dedicated field rather than a third `notes` string on purpose: `notes` already carries two
5865
- * unrelated facts (the user's own note / NO_EXTRACTABLE_TEXT, and the kill-switch markers), so every
5866
- * writer of it clobbers the others.
6579
+ * A dedicated field on purpose, and the precedent #2016 followed for the other two machine-written
6580
+ * facts: while they all shared `notes` every writer of that field clobbered the others, including
6581
+ * the user's own note.
5867
6582
  *
5868
6583
  * Cleared by `commitFabFileChunks` (the rebuild landed) and by the chunk handler's pause write (the
5869
6584
  * rebuild was halted instead). A file carrying `error` is settled regardless - see
@@ -5872,20 +6587,6 @@ const CONVERGENCE_PAUSED_CHUNK_NOTE = "Re-chunking paused by the data-lake conve
5872
6587
  function isChunkRebuildPending(requestedAt) {
5873
6588
  return requestedAt !== null && requestedAt !== void 0 && requestedAt !== "";
5874
6589
  }
5875
- /**
5876
- * Whether a file's `notes` marks it as stalled by the convergence kill switch, by either arm.
5877
- * THE predicate every reader uses, so adding a third stall marker reaches health, convergence and
5878
- * retrieval without three separate string comparisons drifting apart.
5879
- */
5880
- function isConvergencePausedNote(notes) {
5881
- return CONVERGENCE_PAUSED_NOTES.includes(notes);
5882
- }
5883
- /**
5884
- * Datastore mirror of `isConvergencePausedNote`, for a Mongo `notes: { $in: [...] }`. Exported so a
5885
- * query and the in-memory predicate cannot drift: adding a third stall marker to this array reaches
5886
- * both. Declared after the two constants it names so the function above can close over it.
5887
- */
5888
- const CONVERGENCE_PAUSED_NOTES = [CONVERGENCE_PAUSED_NOTE, CONVERGENCE_PAUSED_CHUNK_NOTE];
5889
6590
  /** Ceiling so "adjustable" cannot mean "unbounded" in either direction. */
5890
6591
  const LAKE_ACCESS_AUDIT_RETENTION_MAX_DAYS = 2555;
5891
6592
  /**
@@ -5917,14 +6618,13 @@ const LAKE_CONFIG_AUDIT_RETENTION_DEFAULT_DAYS = 1095;
5917
6618
  /** Ceiling so "adjustable" cannot mean "unbounded" in either direction. */
5918
6619
  const LAKE_CONFIG_AUDIT_RETENTION_MAX_DAYS = 3650;
5919
6620
  /**
5920
- * Forced-retrieval budget defaults, shared between the admin-settings schema in this package and
5921
- * `ChatCompletionFeatures.ts` (which cannot import from `common`'s settings schema without a
5922
- * dependency cycle, so the constant lives here instead).
6621
+ * Forced-retrieval budget and relevance defaults, shared between the admin-settings schema in this
6622
+ * package and `ChatCompletionFeatures.ts` (which cannot import from `common`'s settings schema
6623
+ * without a dependency cycle, so the constants live here instead).
5923
6624
  *
5924
- * Only `FORCED_RETRIEVAL_CHAR_BUDGET_DEFAULT` is a lever (see the `forcedRetrievalCharBudget`
5925
- * setting) - the char budget is the measured binding constraint on how much of a corpus reaches
5926
- * the model on every Data-Lake-mode turn. The relevance floor is exported alongside it so the two
5927
- * stay next to each other, not because it is tunable today.
6625
+ * All three are levers. The char budget is the measured binding constraint on how much of a corpus
6626
+ * reaches the model on every Data-Lake-mode turn; the two floors decide which passages are eligible
6627
+ * to spend it (see `forcedRetrievalRelativeFloorPct` and `forcedRetrievalMinSimilarityPct`).
5928
6628
  */
5929
6629
  /** Total characters of retrieved chunk text injected into a forced-retrieval prompt. */
5930
6630
  const FORCED_RETRIEVAL_CHAR_BUDGET_DEFAULT = 12e3;
@@ -5958,12 +6658,23 @@ let OllamaEmbeddingModel = /* @__PURE__ */ function(OllamaEmbeddingModel) {
5958
6658
  return OllamaEmbeddingModel;
5959
6659
  }({});
5960
6660
  /**
5961
- * The default embedding model for the current deployment. On self-host with a local Ollama
5962
- * server and no cloud embedding key, this is a local embedder, so RAG / knowledge search work
5963
- * out of the box with no AWS/OpenAI/Voyage credential; otherwise it is the cloud default. Read
5964
- * by the `defaultEmbeddingModel` admin-setting default and by the query-embedding fallback, so
5965
- * an operator who never opens admin settings still gets a working, keyless embedder instead of
5966
- * an unconfigured cloud model that fails with an opaque "security token" error.
6661
+ * The default embedding model this deployment advertises - the `defaultEmbeddingModel` admin-setting
6662
+ * default and the query-embedding fallback, so an operator who never opens admin settings still gets
6663
+ * a sensible model rather than an unconfigured one.
6664
+ *
6665
+ * Deliberately answers from the ENVIRONMENT only, and deliberately does NOT try to answer "can this
6666
+ * deployment actually reach that model". It cannot: on a hosted stage an SST secret arrives as a
6667
+ * linked Resource, never as process.env (see apps/client/server/modelDiscovery/adapters.ts), so an
6668
+ * absent OPENAI_API_KEY here means "unknown", not "no key" - it is absent on production too. This
6669
+ * function is also bundled into the browser via settingsMap, where every env read is undefined, so
6670
+ * any answer it gives must be one that is safe as a stage-neutral constant.
6671
+ *
6672
+ * Reachability is therefore decided where the resolved credentials are actually in hand, by
6673
+ * `resolveEmbeddingWithKeylessFallback` (fab-pipeline) - that is what lets a keyless cloud stage
6674
+ * embed on Bedrock without changing what any other stage advertises.
6675
+ *
6676
+ * The one arm that IS env-decidable: a self-host running Ollama has its base URL in real process.env
6677
+ * on both server and (absent) client, and no cloud credential can arrive by any other route.
5967
6678
  */
5968
6679
  function defaultEmbeddingModelForEnv() {
5969
6680
  const selfHost = process.env.B4M_SELF_HOST === "true";
@@ -5972,6 +6683,45 @@ function defaultEmbeddingModelForEnv() {
5972
6683
  if (selfHost && hasOllama && !hasCloudEmbeddingKey) return "qwen3-embedding:0.6b";
5973
6684
  return "text-embedding-ada-002";
5974
6685
  }
6686
+ /**
6687
+ * True when this deployment can embed with no provider API key at all: a cloud stage reaches
6688
+ * Bedrock through its task/execution role's AWS credentials.
6689
+ *
6690
+ * Requires POSITIVE evidence of an execution role rather than merely "not self-host". A plain
6691
+ * `next dev` session and a CI job both leave B4M_SELF_HOST unset while holding no AWS credentials
6692
+ * at all, so an absence test would send them to the Bedrock SDK for an opaque `CredentialsProvider
6693
+ * Error` in place of the actionable OPENAI_KEY_MISSING_MESSAGE naming the key to set - the same
6694
+ * actionable-to-opaque trade this fallback exists to avoid, just in a different keyless place.
6695
+ *
6696
+ * BOTH runtimes must be covered, and they carry different markers. The Lambdas (vectorize
6697
+ * subscriber, the crons, the Next API routes) get AWS_LAMBDA_FUNCTION_NAME; ChatCompletion is a
6698
+ * Fargate service (infra/chatCompletion.ts) and gets the ECS task-role URI instead. Since
6699
+ * knowledgeBaseSearch runs inside that container, keying on the Lambda marker alone would leave
6700
+ * chat knowledge-base search failing on exactly the keyless stages this fallback is for.
6701
+ *
6702
+ * SST_RESOURCE_App is the third arm and the one this repo can prove: SST sets it on anything it
6703
+ * links, Lambda and Service alike (infra/chatCompletion.ts:129 and infra/agentExecutor.ts:54 both
6704
+ * note that linking alone exposes SST_RESOURCE_*). It covers the Fargate task whether or not the
6705
+ * ECS credential URI is present, and it is absent from a plain `next dev` and from CI, which is
6706
+ * the case that matters. `sst dev` does set it - correctly, since that session runs against real
6707
+ * AWS credentials.
6708
+ *
6709
+ * Self-host is excluded outright because it has no such role - its keyless path is the local
6710
+ * Ollama embedder (`isLocalEmbedderAvailable` in toolAvailability.ts), not Bedrock.
6711
+ *
6712
+ * Answers "is Bedrock reachable here", NOT "should we use it" - a keyed stage is keyless-capable
6713
+ * too, so this must only ever be asked ALONGSIDE a resolved credential table that came back empty.
6714
+ * `resolveEmbeddingWithKeylessFallback` is where the two questions are paired for callers free to
6715
+ * choose the model, and is what such a caller should use instead of asking this directly. The
6716
+ * direct callers are the ones that additionally need the answer BEFORE resolving, to decide
6717
+ * policy: toolAvailability reports whether embedding-backed tools are usable at all, and
6718
+ * data-lakes/semantic-search decides whether a substitution is permitted for this request before
6719
+ * it knows whether one is needed. Both still pair it with the table.
6720
+ */
6721
+ function hasKeylessCloudEmbedder() {
6722
+ if (process.env.B4M_SELF_HOST === "true") return false;
6723
+ return !!(process.env.AWS_LAMBDA_FUNCTION_NAME || process.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI || process.env.AWS_CONTAINER_CREDENTIALS_FULL_URI || process.env.SST_RESOURCE_App);
6724
+ }
5975
6725
  z$1.union([
5976
6726
  z$1.enum(OpenAIEmbeddingModel),
5977
6727
  z$1.enum(VoyageAIEmbeddingModel),
@@ -5979,6 +6729,18 @@ z$1.union([
5979
6729
  z$1.enum(OllamaEmbeddingModel)
5980
6730
  ]);
5981
6731
  /**
6732
+ * The measured per-space floors, rendered for an admin-facing description (e.g. "75 for
6733
+ * text-embedding-ada-002, 35 for text-embedding-3-small").
6734
+ *
6735
+ * Rendered rather than written out in prose because these numbers are expected to move - 35 is
6736
+ * provisional until it is re-derived against a production lake - and a description that restates
6737
+ * the table is a wrong number shown to operators the moment it drifts, with nothing failing.
6738
+ */
6739
+ const forcedRetrievalFloorsBySpaceSummary = Object.entries({
6740
+ ["text-embedding-ada-002"]: 75,
6741
+ ["text-embedding-3-small"]: 35
6742
+ }).map(([space, pct]) => `${pct} for ${space}`).join(", ");
6743
+ /**
5982
6744
  * Default text for the artifact-emission system prompt. Single source of truth used BOTH as the
5983
6745
  * `ArtifactEmissionPrompt` admin setting's default AND as the runtime fallback in
5984
6746
  * ChatCompletionProcess - so an unset/empty/cleared DB value reverts to this and never bricks
@@ -6082,6 +6844,76 @@ const HELP_CENTER_PROMPT = `HELP CENTER: Bike4Mind has a built-in Help Center th
6082
6844
  */
6083
6845
  const ABSTENTION_PROMPT = `When a request is underspecified or your sources do not cover it, say so and name what is missing. "I do not have enough to answer that" is a correct, high-value answer. Never invent facts about the user, their business, or their data, and never state a specific customer, competitor, deal, or figure as fact - or cite a source for it - unless your sources support it, even when the question assumes it.`;
6084
6846
  /**
6847
+ * Default text for the web-search freshness nudge, and the `WebSearchFreshnessPrompt` admin
6848
+ * setting's default.
6849
+ *
6850
+ * Unlike ABSTENTION_PROMPT / ARTIFACT_EMISSION_PROMPT / HELP_CENTER_PROMPT, this setting
6851
+ * distinguishes an absent row from a cleared one. ChatCompletionProcess reads it 2-arg, so an
6852
+ * absent row still falls back to this constant as the setting's registered default, but a cleared
6853
+ * '' is returned verbatim and drops the section rather than reverting. The siblings are read 3-arg
6854
+ * and collapse both cases to the constant. That divergence is deliberate - this section has no
6855
+ * companion boolean, so clearing the field is the only off switch it has. Keep the setting's
6856
+ * description in sync with that if either changes.
6857
+ *
6858
+ * Names no tool but `web_search`: the section is gated on web_search being offered, and web_fetch
6859
+ * is an independent toggle that may well be off.
6860
+ */
6861
+ const WEB_SEARCH_FRESHNESS_PROMPT = `# WEB SEARCH AND FRESHNESS
6862
+
6863
+ Your training data has a cutoff. The current date is supplied to you in this conversation's system context - treat it as authoritative, and assume anything time-sensitive may have changed since your training.
6864
+
6865
+ Call \`web_search\` BEFORE answering when the answer depends on a fact that changes over time: current prices or rates, product availability or roadmap status, funding, organizational or personnel changes, published benchmarks or performance figures, competitive positioning, or anything the user frames as "current", "latest", "now", or "as of today". When a stale answer would mislead, search instead of answering from memory. When a search surfaces a specific page that matters, or the user names one, read that page directly rather than answering from the snippet.
6866
+
6867
+ You do not need to search for stable knowledge (definitions, mathematics, established theory), or for questions answerable purely from this conversation or from documents already retrieved for you.
6868
+
6869
+ When you report a time-sensitive fact, state what it is as of - the date of the source you used - and say plainly when you could not verify something and are answering from training data instead. Never present an unverified recollection as a current fact.`;
6870
+ /**
6871
+ * Default text for the knowledge-base retrieval nudge, and the `KnowledgeBaseRetrievalPrompt`
6872
+ * admin setting's default.
6873
+ *
6874
+ * The gap this closes: the tool prompt has a when-to-use section for the clock, for web search,
6875
+ * for MCP and for agent delegation, and none for the user's own corpus. The
6876
+ * `search_knowledge_base` description is entirely HOW to search ("Make ONE good search per
6877
+ * distinct topic") and never WHEN, so on the optional path the model decides unaided - and over 30
6878
+ * days of production it reached for the corpus on 20.1% of the turns it was offered on.
6879
+ *
6880
+ * Read 2-arg by ChatCompletionProcess, exactly as WEB_SEARCH_FRESHNESS_PROMPT is and unlike the
6881
+ * 3-arg siblings: an absent row falls back to this constant as the registered default, but a
6882
+ * cleared '' is returned verbatim and drops the section instead of reverting. Deliberate - the
6883
+ * section has no companion boolean, so clearing the field is its only off switch, and that off
6884
+ * switch is what makes it A/B-able without a deploy. Keep the setting's description in sync.
6885
+ *
6886
+ * Names no tool but `search_knowledge_base`, for the same reason the web-search section names no
6887
+ * `web_fetch`: the companion `retrieve_knowledge_content` is paired in at build time but a session
6888
+ * denylist can still strip it (ChatCompletionProcess warns on exactly that case), and instructing
6889
+ * the model to call a tool it was not given makes it emit the call as leaked JSON text.
6890
+ *
6891
+ * The "do not search" paragraph is load-bearing, not padding. A when-to-retrieve nudge without a
6892
+ * don't-retrieve clause buys retrieval on turns that need none - the same failure mode global
6893
+ * forced retrieval already shows on out-of-corpus questions, reached by a different route. Three of
6894
+ * its clauses are load-bearing for a specific co-resident path, not general hedging:
6895
+ * - "from an attached document" - a small attached corpus is INLINED rather than deferred to
6896
+ * retrieval (`shouldDeferCorpusToRetrieval`), and forced retrieval deliberately steps aside on
6897
+ * an attached-files turn (`forcedRetrievalAbstention` emits nothing there). Without this clause
6898
+ * the section tells the model to go searching for content already sitting in its context.
6899
+ * - "already been searched on this turn" - on a forced turn that found nothing,
6900
+ * `forcedRetrievalNoContextPrompt` instructs the model to say the library does not cover the
6901
+ * question. A nudge to search then invites a second identical query - a billed query embedding,
6902
+ * and a chance to talk itself out of a correct abstention.
6903
+ * - the opening scope, "unless its content has been placed in this conversation" - the reason the
6904
+ * first paragraph does not simply claim the documents are invisible, which is false whenever a
6905
+ * corpus was inlined.
6906
+ */
6907
+ const KNOWLEDGE_BASE_RETRIEVAL_PROMPT = `# KNOWLEDGE BASE
6908
+
6909
+ \`search_knowledge_base\` searches a library of documents the user has made available to you - their own uploads, and any shared or organization library they can reach. You cannot see what a document holds unless its content has been placed in this conversation or you search for it; file names and tags are labels, not content.
6910
+
6911
+ Call \`search_knowledge_base\` BEFORE answering when that library would settle the question: anything about their organization, projects, customers, products, processes or people; a term, name, acronym or identifier that is not general public knowledge; a policy, decision, figure or date specific to them; or a question that assumes context this conversation never gave you. If you are about to answer in general terms a question the user means specifically, search first. A general-knowledge answer that sounds right is the failure this library exists to prevent.
6912
+
6913
+ Do not search when the answer is already in front of you or out of scope: general knowledge (definitions, mathematics, established theory, public facts); anything answerable from this conversation, from an attached document, or from content already retrieved for you this turn; or a request to transform, summarize or reformat text the user has just supplied. If the library has already been searched on this turn, do not search it again for the same question - a repeat spends a round trip to return the same passages.
6914
+
6915
+ When a search does not turn up what was asked for, say so plainly rather than filling the gap from training data, and never imply an answer came from the user's documents when it did not.`;
6916
+ /**
6085
6917
  * Default text for the formatting system message. Runtime fallback used by
6086
6918
  * `includeHardcodedSystemMessage` (b4m-core/utils/src/llm/utils.ts) when the `FormatPromptTemplate`
6087
6919
  * admin setting is blank; that setting's own default is intentionally '' - keep this the sole home.
@@ -6101,6 +6933,7 @@ z$1.enum([
6101
6933
  "geminiDemoKey",
6102
6934
  "xaiApiKey",
6103
6935
  "moonshotApiKey",
6936
+ "deepseekApiKey",
6104
6937
  "voyageApiKey",
6105
6938
  "FirecrawlApiKey",
6106
6939
  "FirecrawlApiUrl",
@@ -6114,6 +6947,8 @@ z$1.enum([
6114
6947
  "ArtifactEmissionPrompt",
6115
6948
  "HelpCenterPrompt",
6116
6949
  "AbstentionPrompt",
6950
+ "WebSearchFreshnessPrompt",
6951
+ "KnowledgeBaseRetrievalPrompt",
6117
6952
  "UseFormatPrompt",
6118
6953
  "EnableQuestMaster",
6119
6954
  "EnableQuestMasterDefault",
@@ -6132,11 +6967,11 @@ z$1.enum([
6132
6967
  "EnableLattice",
6133
6968
  "EnableLatticeDefault",
6134
6969
  "EnableDataLakes",
6135
- "EnableDataLakesDefault",
6136
6970
  "EnableDataLakeSlackAdd",
6137
6971
  "EnableDataLakeGroundingMode",
6138
6972
  "EnableLakeMemory",
6139
6973
  "EnableDataLakeVectorSearch",
6974
+ "EnableRetrievalSupersessionCollapse",
6140
6975
  "PauseLakeConvergence",
6141
6976
  "LakeConvergenceBulkChangeSharePct",
6142
6977
  "EnforceLakeReadGrants",
@@ -6233,16 +7068,21 @@ z$1.enum([
6233
7068
  "defaultEmbeddingModel",
6234
7069
  "dataLakeSearchMaxFiles",
6235
7070
  "dataLakeSearchMaxChunks",
7071
+ "dataLakeSearchMaxChunksPerFile",
6236
7072
  "forcedRetrievalCharBudget",
7073
+ "lakeMemoryRecallK",
6237
7074
  "kbSearchDefaultResults",
6238
7075
  "kbSearchResultTokenBudget",
6239
7076
  "kbSearchMinRelevancePct",
7077
+ "forcedRetrievalRelativeFloorPct",
7078
+ "forcedRetrievalMinSimilarityPct",
6240
7079
  "dataLakeEmbeddingSpendEnabled",
6241
7080
  "dataLakeEmbeddingBudgetPerRunUsd",
6242
7081
  "dataLakeEmbeddingBudgetPerLakeUsd",
6243
7082
  "dataLakeEmbeddingBudgetPerPeriodUsd",
6244
7083
  "dataLakeEmbeddingBudgetPeriodHours",
6245
7084
  "dataLakeEmbeddingMaxCallsPerMinute",
7085
+ "dataLakeEmbeddingMaxTokensPerMinute",
6246
7086
  "dataLakeVectorizeChunkBatchSize",
6247
7087
  "dataLakeEmbeddingTierMultiplierIndividual",
6248
7088
  "dataLakeEmbeddingTierMultiplierOrganization",
@@ -6291,7 +7131,6 @@ z$1.enum([
6291
7131
  "EnableBmPiDefault",
6292
7132
  "EnableBmPiJira",
6293
7133
  "EnableOptiHashi",
6294
- "EnableOptiHashiDefault",
6295
7134
  "EnableComputeSubmission",
6296
7135
  "EnableFamilyCompute",
6297
7136
  "EnableHybridCompute",
@@ -6319,6 +7158,7 @@ z$1.enum([
6319
7158
  "modelDiscoveryAllowEgress",
6320
7159
  "modelDiscoveryPriceBandPct",
6321
7160
  "modelDiscoveryAutoRemap",
7161
+ "modelDiscoveryProbeNewModels",
6322
7162
  "prReportRepo",
6323
7163
  "prReportIdentityMap",
6324
7164
  "prReportWebhookUrl",
@@ -6361,10 +7201,12 @@ const OrchestrationDefaultsSchema = z$1.object({
6361
7201
  "mermaid_chart"
6362
7202
  ]),
6363
7203
  /**
6364
- * Tool names explicitly forbidden. Enforced as a final subtraction in
6365
- * `pickEffectiveEnabledTools` - wins even over payload-pinned tools - so this
6366
- * is the defense-in-depth backstop for the case where an admin broadens
6367
- * `allowedTools` without realizing a parallel denylist is also needed.
7204
+ * Tool names explicitly forbidden. Enforced in two places: as a final subtraction in
7205
+ * `pickEffectiveEnabledTools` (wins even over payload-pinned tools), and - for the two
7206
+ * delegation tools, which are injected as objects and never registered by name - at the
7207
+ * dependency gate in agentExecutor (`delegationOffer` withholds `agentStore` /
7208
+ * `dagDispatcher`). The name subtraction alone cannot reach those two; see
7209
+ * agentExecutor.sessionToolPolicy.
6368
7210
  *
6369
7211
  * Seeded with every tool that mutates user data (the spec's
6370
7212
  * "anything tagged `mutates_user_data`"): destructive/overwriting filesystem
@@ -6442,8 +7284,30 @@ const DATA_LAKE_SEARCH_MAX_CHUNKS_DEFAULT = 1e5;
6442
7284
  const DATA_LAKE_EMBEDDING_BUDGET_PER_LAKE_USD_MAX = 1e4;
6443
7285
  const DATA_LAKE_EMBEDDING_BUDGET_PER_PERIOD_USD_MAX = 5e3;
6444
7286
  const DATA_LAKE_EMBEDDING_MAX_CALLS_PER_MINUTE_MAX = 1e4;
7287
+ /**
7288
+ * The TOKEN half of the throughput cap, and the one that maps to what providers actually meter.
7289
+ * A call cap alone does not bound tokens: one call carries up to
7290
+ * DATA_LAKE_VECTORIZE_CHUNK_BATCH_SIZE_DEFAULT passages of DEFAULT_PASSAGE_TOKEN_TARGET tokens, so
7291
+ * 120 calls/min permits ~3.1M tokens/min - several times the smallest paid embeddings tier. The two
7292
+ * levers are complementary: calls/min bounds RPM, this bounds TPM, and a call must fit both.
7293
+ *
7294
+ * The default is deliberately LOW - it has to be safe on the smallest tier any deployment might
7295
+ * be on, including self-hosts nobody here can see. It is not a claim about what any particular
7296
+ * account can do, and reading it as one is the mistake to avoid: a provider tier is a property of
7297
+ * the provider organization, so it cannot be derived from this codebase at all.
7298
+ *
7299
+ * The real number is measurable per deployment: Admin -> Settings -> AI -> Data Lake Cost
7300
+ * Governance reads the configured provider's live ceiling (GET /api/admin/embedding-limits) and
7301
+ * shows it beside this lever, so an operator sets this from their own measured quota rather than
7302
+ * from a guess baked in here. Leave headroom below the measured ceiling for QUERY-side embedding,
7303
+ * which is exempt from this gate (see enforceEmbeddingSpendGate) and shares the same per-model
7304
+ * pool - a retrieval query must not queue behind a backfill.
7305
+ */
7306
+ const DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_DEFAULT = 6e5;
7307
+ const DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_MAX = 5e7;
6445
7308
  function makeNumberSetting(config) {
6446
7309
  let numberSchema = z$1.coerce.number();
7310
+ if (config.int) numberSchema = numberSchema.int();
6447
7311
  if (config.min !== void 0) numberSchema = numberSchema.min(config.min);
6448
7312
  if (config.max !== void 0) numberSchema = numberSchema.max(config.max);
6449
7313
  return {
@@ -7010,6 +7874,14 @@ const API_SERVICE_GROUPS = {
7010
7874
  {
7011
7875
  key: "AbstentionPrompt",
7012
7876
  order: 11
7877
+ },
7878
+ {
7879
+ key: "WebSearchFreshnessPrompt",
7880
+ order: 12
7881
+ },
7882
+ {
7883
+ key: "KnowledgeBaseRetrievalPrompt",
7884
+ order: 13
7013
7885
  }
7014
7886
  ]
7015
7887
  },
@@ -7046,6 +7918,22 @@ const API_SERVICE_GROUPS = {
7046
7918
  {
7047
7919
  key: "kbSearchMinRelevancePct",
7048
7920
  order: 7
7921
+ },
7922
+ {
7923
+ key: "lakeMemoryRecallK",
7924
+ order: 8
7925
+ },
7926
+ {
7927
+ key: "forcedRetrievalRelativeFloorPct",
7928
+ order: 9
7929
+ },
7930
+ {
7931
+ key: "forcedRetrievalMinSimilarityPct",
7932
+ order: 10
7933
+ },
7934
+ {
7935
+ key: "dataLakeSearchMaxChunksPerFile",
7936
+ order: 11
7049
7937
  }
7050
7938
  ]
7051
7939
  },
@@ -7080,16 +7968,20 @@ const API_SERVICE_GROUPS = {
7080
7968
  order: 6
7081
7969
  },
7082
7970
  {
7083
- key: "dataLakeVectorizeChunkBatchSize",
7971
+ key: "dataLakeEmbeddingMaxTokensPerMinute",
7084
7972
  order: 7
7085
7973
  },
7086
7974
  {
7087
- key: "dataLakeEmbeddingTierMultiplierIndividual",
7975
+ key: "dataLakeVectorizeChunkBatchSize",
7088
7976
  order: 8
7089
7977
  },
7090
7978
  {
7091
- key: "dataLakeEmbeddingTierMultiplierOrganization",
7979
+ key: "dataLakeEmbeddingTierMultiplierIndividual",
7092
7980
  order: 9
7981
+ },
7982
+ {
7983
+ key: "dataLakeEmbeddingTierMultiplierOrganization",
7984
+ order: 10
7093
7985
  }
7094
7986
  ]
7095
7987
  },
@@ -7149,6 +8041,16 @@ const API_SERVICE_GROUPS = {
7149
8041
  order: 1
7150
8042
  }]
7151
8043
  },
8044
+ DEEPSEEK: {
8045
+ id: "deepseekAPIService",
8046
+ name: "DeepSeek Service",
8047
+ description: "DeepSeek API integration settings",
8048
+ icon: "AutoAwesome",
8049
+ settings: [{
8050
+ key: "deepseekApiKey",
8051
+ order: 1
8052
+ }]
8053
+ },
7152
8054
  ANTHROPIC: {
7153
8055
  id: "anthropicAPIService",
7154
8056
  name: "Anthropic Service",
@@ -7486,10 +8388,6 @@ const API_SERVICE_GROUPS = {
7486
8388
  key: "EnableOptiHashi",
7487
8389
  order: 80
7488
8390
  },
7489
- {
7490
- key: "EnableOptiHashiDefault",
7491
- order: 81
7492
- },
7493
8391
  {
7494
8392
  key: "EnableComputeSubmission",
7495
8393
  order: 82
@@ -7852,6 +8750,10 @@ const API_SERVICE_GROUPS = {
7852
8750
  {
7853
8751
  key: "modelDiscoveryAutoRemap",
7854
8752
  order: 6
8753
+ },
8754
+ {
8755
+ key: "modelDiscoveryProbeNewModels",
8756
+ order: 7
7855
8757
  }
7856
8758
  ]
7857
8759
  },
@@ -7937,6 +8839,16 @@ const settingsMap = {
7937
8839
  group: API_SERVICE_GROUPS.MOONSHOT.id,
7938
8840
  order: 1
7939
8841
  }),
8842
+ deepseekApiKey: makeStringSetting({
8843
+ key: "deepseekApiKey",
8844
+ name: "DeepSeek API Key",
8845
+ defaultValue: "",
8846
+ description: "The global API Key for DeepSeek.",
8847
+ isSensitive: true,
8848
+ category: "AI",
8849
+ group: API_SERVICE_GROUPS.DEEPSEEK.id,
8850
+ order: 1
8851
+ }),
7940
8852
  voyageApiKey: makeStringSetting({
7941
8853
  key: "voyageApiKey",
7942
8854
  name: "Voyage API Key",
@@ -7985,16 +8897,6 @@ const settingsMap = {
7985
8897
  group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
7986
8898
  order: 88
7987
8899
  }),
7988
- EnableDataLakesDefault: makeBooleanSetting({
7989
- key: "EnableDataLakesDefault",
7990
- name: "Data Lakes: On by default for users",
7991
- defaultValue: false,
7992
- description: "When enabled, Data Lakes is active for users who have never explicitly toggled it.",
7993
- category: "Experimental",
7994
- group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
7995
- order: 89,
7996
- dependsOn: "EnableDataLakes"
7997
- }),
7998
8900
  EnableDataLakeSlackAdd: makeBooleanSetting({
7999
8901
  key: "EnableDataLakeSlackAdd",
8000
8902
  name: "Data Lakes: Slack \"@datalake add\" path",
@@ -8019,7 +8921,7 @@ const settingsMap = {
8019
8921
  key: "EnableLakeMemory",
8020
8922
  name: "Data Lakes: Lake memory profile (extraction)",
8021
8923
  defaultValue: false,
8022
- description: "Server-side gate for the lake memory producer - LLM extraction of a data lake's documents into a durable memory profile on ingest. Off by default (measurement rollout); the consumer that injects the profile is inert until this is on and a lake has been extracted.",
8924
+ description: "Master gate for lake memory, on all three sides: LLM extraction of a data lake's documents into a durable memory profile, recall of that profile into chats grounded in the lake, and whether the per-lake opt-in is offered at all. Off by default (measurement rollout). Turning it off stops recall immediately and stops new extractions from being queued or picked up, though a run already in flight finishes its slice (bounded by the handler timeout). It is NOT destructive - each lake keeps its own opt-in and its built profile, so flipping this back on resumes where it left off. Erasing a profile is a separate, explicit per-lake action.",
8023
8925
  category: "Experimental",
8024
8926
  group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8025
8927
  order: 91,
@@ -8035,11 +8937,21 @@ const settingsMap = {
8035
8937
  order: 92,
8036
8938
  dependsOn: "EnableDataLakes"
8037
8939
  }),
8940
+ EnableRetrievalSupersessionCollapse: makeBooleanSetting({
8941
+ key: "EnableRetrievalSupersessionCollapse",
8942
+ name: "Data Lakes: Collapse superseded members before ranking",
8943
+ defaultValue: false,
8944
+ description: "When a lake holds two generations of the same document (a re-upload, a Drive sync, a migration), rank only the newest and report the suppression. Off by default: the weakest identity tier is a bare file name, so two genuinely different documents sharing a name in one lake would collapse to one - turn this on only after checking the reported collapse counts on real lakes. Suppression is recoverable either way; a collapsed member is still reachable by id or name through retrieve_knowledge_content.",
8945
+ category: "Experimental",
8946
+ group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8947
+ order: 97,
8948
+ dependsOn: "EnableDataLakes"
8949
+ }),
8038
8950
  PauseLakeConvergence: makeBooleanSetting({
8039
8951
  key: "PauseLakeConvergence",
8040
8952
  name: "Data Lakes: Pause background convergence work",
8041
8953
  defaultValue: false,
8042
- description: "Kill switch for background data-lake ingestion work (convergence sweeps, rescue re-chunking) - NOT real-time user uploads, which are always honored. Off by default. Turn ON to halt in-flight background chunk/vectorize messages the next time the handler picks them up (a re-check inside the shared handler, so it takes effect on work already queued, not just the next scheduling pass). The platform value pauses every lake at once; a per-lake (or per-org / per-owner) override pauses a subset while the rest keep running. A platform-level flip applies immediately to lake-wide work and within ~5 min to per-lake-scoped work (settings cache).",
8954
+ description: "Kill switch for background data-lake ingestion work (convergence sweeps, rescue re-chunking) - NOT real-time user uploads, which are always honored. Off by default. Turn ON to halt in-flight background chunk/vectorize messages the next time the handler picks them up (a re-check inside the shared handler, so it takes effect on work already queued, not just the next scheduling pass). The platform value pauses every lake at once; a per-lake (or per-org / per-owner) override pauses a subset while the rest keep running - including overriding a platform-wide pause back OFF for one lake. Every producer honors the override, the global chunk rescue sweep included: it resolves each candidate against the lake it belongs to (#2157), so a file in no lake at all follows the platform value, which is correct for it. A platform-level flip applies immediately to lake-wide work and within ~5 min to per-lake-scoped work (settings cache).",
8043
8955
  category: "Experimental",
8044
8956
  group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8045
8957
  order: 93,
@@ -8069,8 +8981,8 @@ const settingsMap = {
8069
8981
  EnforceLakeReadGrants: makeBooleanSetting({
8070
8982
  key: "EnforceLakeReadGrants",
8071
8983
  name: "Data Lakes: Enforce read-time grant resolution",
8072
- defaultValue: false,
8073
- description: "Read-time grant cutover (#1673). OFF by default = report-only: the read gate resolves a persisted READER/org grant into an ephemeral membership view and logs where it WOULD change access ([lakeReadGrantCutover] lines), but the enforced decision stays the legacy owner/org/tag/entitlement/public rule so no one gains or loses access. NOTE: turning this ON is currently a NO-OP guarded by a source-level interlock (READ_GRANT_ENFORCEMENT_READY) - enforcement will not activate until the follow-up code (member-management write path + retrieval arm) lands and flips it, and a premature toggle just logs a warning and stays report-only. This is deliberate so the setting cannot half-enable a half-wired gate. Platform altitude on purpose: a one-time install-wide migration cutover, not a per-lake lever. Tag and entitlement grants always resolve live and are never affected by this flag; only persisted reader/org rows are gated by it.",
8984
+ defaultValue: true,
8985
+ description: "Read-time grant resolution (#1673). ON is the shipped default: a persisted READER or ORG grant is resolved into the read decision, so a principal a lake was shared with can browse it, open it and ground on it. Resolution is purely ADDITIVE (legacy OR grant), so it takes no access away; this arm contains an ORG grant to the granting org, and expired rows never resolve. This is the standing KILL SWITCH for that arm, not a migration phase: turning it OFF returns to report-only, where the gate still resolves grants and logs where they WOULD change access ([lakeReadGrantCutover] lines) but the enforced decision falls back to the legacy owner/org/tag/entitlement/public rule - so those log lines are the diagnostic for a lake someone can no longer reach while the switch is off. Platform altitude on purpose: install-wide, not a per-lake lever. Tag and entitlement grants always resolve live and are never affected by this flag; only persisted reader/org rows are gated by it.",
8074
8986
  category: "Experimental",
8075
8987
  group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
8076
8988
  order: 94,
@@ -8320,6 +9232,7 @@ const settingsMap = {
8320
9232
  }),
8321
9233
  DefaultChunkSize: makeNumberSetting({
8322
9234
  key: "DefaultChunkSize",
9235
+ userReadable: true,
8323
9236
  name: "Default Chunk Size",
8324
9237
  defaultValue: 512,
8325
9238
  min: 64,
@@ -8378,6 +9291,22 @@ const settingsMap = {
8378
9291
  category: "AI",
8379
9292
  order: 11
8380
9293
  }),
9294
+ WebSearchFreshnessPrompt: makeStringSetting({
9295
+ key: "WebSearchFreshnessPrompt",
9296
+ name: "Web Search Freshness Prompt",
9297
+ defaultValue: WEB_SEARCH_FRESHNESS_PROMPT,
9298
+ description: "System prompt telling the model when to reach for web_search rather than answer from training data, and to state the as-of date of any time-sensitive fact. Injected only when the web_search tool is offered for the request - a model instructed to search without a search tool tends to claim it searched. Clearing this field turns the section OFF rather than restoring the built-in default, and it is the only off switch this section has; to get the stock wording back, paste it in. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else. After an upgrade, diff a saved copy against the built-in default: a saved copy pins the wording from whenever it was saved and will not pick up fixes made since.",
9299
+ category: "AI",
9300
+ order: 12
9301
+ }),
9302
+ KnowledgeBaseRetrievalPrompt: makeStringSetting({
9303
+ key: "KnowledgeBaseRetrievalPrompt",
9304
+ name: "Knowledge Base Retrieval Prompt",
9305
+ defaultValue: KNOWLEDGE_BASE_RETRIEVAL_PROMPT,
9306
+ description: "System prompt telling the model when to reach for search_knowledge_base rather than answer from training data, and when NOT to. Injected only when the search_knowledge_base tool is offered for the request - a model instructed to search a corpus it has no tool for tends to claim it searched. Clearing this field turns the section OFF rather than restoring the built-in default, and it is the only off switch this section has; to get the stock wording back, paste it in. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else. After an upgrade, diff a saved copy against the built-in default: a saved copy pins the wording from whenever it was saved and will not pick up fixes made since.",
9307
+ category: "AI",
9308
+ order: 13
9309
+ }),
8381
9310
  UseFormatPrompt: makeBooleanSetting({
8382
9311
  key: "UseFormatPrompt",
8383
9312
  name: "Use Format Prompt",
@@ -8396,6 +9325,7 @@ const settingsMap = {
8396
9325
  }),
8397
9326
  pricePerCredit: makeNumberSetting({
8398
9327
  key: "pricePerCredit",
9328
+ userReadable: true,
8399
9329
  name: "Price Per Credit",
8400
9330
  defaultValue: 50,
8401
9331
  description: "The price per credit for purchasing credits.",
@@ -8473,7 +9403,8 @@ const settingsMap = {
8473
9403
  name: "Referal Credits Amount",
8474
9404
  defaultValue: 1e4,
8475
9405
  description: "Credits to give to the referred user.",
8476
- category: "Referrals"
9406
+ category: "Referrals",
9407
+ userReadable: true
8477
9408
  }),
8478
9409
  EnableReferralToEmail: makeBooleanSetting({
8479
9410
  key: "EnableReferralToEmail",
@@ -8662,8 +9593,10 @@ const settingsMap = {
8662
9593
  }),
8663
9594
  MaxFileSize: makeNumberSetting({
8664
9595
  key: "MaxFileSize",
9596
+ userReadable: true,
8665
9597
  name: "Max File Size",
8666
9598
  defaultValue: 30,
9599
+ min: 1,
8667
9600
  description: "The maximum file size allowed for uploads in MB.",
8668
9601
  category: "Knowledge",
8669
9602
  group: API_SERVICE_GROUPS.KNOWLEDGE.id,
@@ -8770,6 +9703,7 @@ const settingsMap = {
8770
9703
  }),
8771
9704
  enforceCredits: makeBooleanSetting({
8772
9705
  key: "enforceCredits",
9706
+ userReadable: true,
8773
9707
  name: "Enforce Credits",
8774
9708
  defaultValue: process.env.B4M_SELF_HOST === "true" ? false : true,
8775
9709
  description: "Whether to enforce credits for users",
@@ -8786,6 +9720,7 @@ const settingsMap = {
8786
9720
  }),
8787
9721
  enableTeamPlan: makeBooleanSetting({
8788
9722
  key: "enableTeamPlan",
9723
+ userReadable: true,
8789
9724
  name: "Enable Team Plan",
8790
9725
  defaultValue: false,
8791
9726
  description: "Whether to enable team plans",
@@ -8891,7 +9826,8 @@ const settingsMap = {
8891
9826
  description: "The global system prompt files to be used for AI model configuration.",
8892
9827
  category: "AI",
8893
9828
  group: API_SERVICE_GROUPS.OPENAI.id,
8894
- order: 8
9829
+ order: 8,
9830
+ userReadable: true
8895
9831
  }),
8896
9832
  OpenWeatherKey: makeStringSetting({
8897
9833
  key: "OpenWeatherKey",
@@ -9045,6 +9981,7 @@ const settingsMap = {
9045
9981
  }),
9046
9982
  MaxContentLength: makeNumberSetting({
9047
9983
  key: "MaxContentLength",
9984
+ userReadable: true,
9048
9985
  name: "Max Content Length",
9049
9986
  defaultValue: 5e4,
9050
9987
  description: "The maximum character length for file content displayed in workbench (truncated if larger).",
@@ -9210,9 +10147,10 @@ const settingsMap = {
9210
10147
  }),
9211
10148
  defaultEmbeddingModel: makeStringSetting({
9212
10149
  key: "defaultEmbeddingModel",
10150
+ userReadable: true,
9213
10151
  name: "Default Embedding Model",
9214
10152
  defaultValue: defaultEmbeddingModelForEnv(),
9215
- description: "The default embedding model to use",
10153
+ description: "The default embedding model to use. Changing it changes the SCALE of every similarity score in the system, so relevance floors do not carry across: a floor tuned for one model can sit above the entire range of another and reject everything. The server handles this for you on the floors it ships, applying the value measured for whichever model your documents are actually embedded with - but if you have set Forced Retrieval Absolute Floor by hand, re-measure it after changing this. Existing documents keep their old vectors and are only comparable to a query embedded the same way, so a change here needs a re-embed to take full effect; until then each set of documents is searched with the model it was indexed under.",
9216
10154
  category: "AI",
9217
10155
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9218
10156
  options: [
@@ -9231,11 +10169,7 @@ const settingsMap = {
9231
10169
  category: "AI",
9232
10170
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9233
10171
  order: 2,
9234
- scope: { settableAt: [
9235
- "organization",
9236
- "owner",
9237
- "lake"
9238
- ] }
10172
+ scope: { settableAt: ["organization", "owner"] }
9239
10173
  }),
9240
10174
  dataLakeSearchMaxChunks: makeNumberSetting({
9241
10175
  key: "dataLakeSearchMaxChunks",
@@ -9246,11 +10180,18 @@ const settingsMap = {
9246
10180
  category: "AI",
9247
10181
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9248
10182
  order: 3,
9249
- scope: { settableAt: [
9250
- "organization",
9251
- "owner",
9252
- "lake"
9253
- ] }
10183
+ scope: { settableAt: ["organization", "owner"] }
10184
+ }),
10185
+ dataLakeSearchMaxChunksPerFile: makeNumberSetting({
10186
+ key: "dataLakeSearchMaxChunksPerFile",
10187
+ name: "Data Lake Search Max Chunks Per Document",
10188
+ defaultValue: 0,
10189
+ min: 0,
10190
+ description: "Most chunks from any ONE source document a data-lake semantic search may return in its top-K. A diversity guard for CONTESTED slots: where several documents answer the question, it stops the best-scoring one from taking slots the others could have filled. It is NOT a fix for severe crowding - the cap redistributes only among the candidates retrieval already returned, so a document that supplies enough of the top-scoring chunks to fill that pool on its own is one the cap cannot change at all. On a corpus of book-length documents, expect enabling this to change little beyond widening the vector-search request. 0 (default) disables the cap, byte-identical to behavior before this setting existed. The cap never SHRINKS a result set - once the spread-out picks are in, any slots still open are backfilled with the highest-scoring chunks the cap held back, so a lake whose only match is one document still returns a full top-K. A value at or above the result count is also a no-op, since nothing can ever be held back. Below it, each retrieval stream's candidate pool is widened to a fixed multiple of the result count so the cap has a spread to choose from. The scanned corpus itself does not grow (that is bounded separately), but the vector-search backends are asked for that many more matches, and a larger in-memory ranking pool costs some CPU. 2-3 is the useful range; 1 serves one passage per document, which suits a corpus of many short documents and starves a question whose answer spans one long one. The chat knowledge-base path ranks more passages than it serves, so it applies the cap a second time at the count it actually serves - otherwise the spread-out picks, which are by definition the lowest-scoring ones admitted, would land in the passages that path discards.",
10191
+ category: "AI",
10192
+ group: API_SERVICE_GROUPS.EMBEDDING.id,
10193
+ order: 11,
10194
+ scope: { settableAt: ["organization", "owner"] }
9254
10195
  }),
9255
10196
  forcedRetrievalCharBudget: makeNumberSetting({
9256
10197
  key: "forcedRetrievalCharBudget",
@@ -9258,10 +10199,11 @@ const settingsMap = {
9258
10199
  defaultValue: FORCED_RETRIEVAL_CHAR_BUDGET_DEFAULT,
9259
10200
  min: 1e3,
9260
10201
  max: 1e5,
9261
- description: "Total characters of retrieved chunk text injected into a Data-Lake-mode turn. Measured saturating on every turn against a 47-document lake, so this is the binding constraint on how much of a corpus reaches the model - not the relevance floor. Raising it admits more passages at the cost of prompt tokens and latency on every Data-Lake turn; it is NOT automatically better, since more context can dilute ranking. Platform-only for now: this read does not go through the scoped-settings resolver, so a `settableAt` block here would be inert metadata at best and could arm the resolver's fail-loud owner check at worst.",
10202
+ description: "Total characters of retrieved chunk text injected into a Data-Lake-mode turn. Measured saturating on every turn against a 47-document lake, so this is the binding constraint on how much of a corpus reaches the model - not the relevance floor. Raising it admits more passages at the cost of prompt tokens and latency on every Data-Lake turn; it is NOT automatically better, since more context can dilute ranking. Overridable per organization and per owner, the same altitude as the two relevance floors resolved alongside it on the same turn.",
9262
10203
  category: "AI",
9263
10204
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9264
- order: 4
10205
+ order: 4,
10206
+ scope: { settableAt: ["organization", "owner"] }
9265
10207
  }),
9266
10208
  kbSearchDefaultResults: makeNumberSetting({
9267
10209
  key: "kbSearchDefaultResults",
@@ -9269,7 +10211,7 @@ const settingsMap = {
9269
10211
  defaultValue: 5,
9270
10212
  min: 1,
9271
10213
  max: 10,
9272
- description: "Passages the search_knowledge_base tool returns when a model call omits max_results, which is most calls. This is the exact bound while kbSearchResultTokenBudget is unset (0). Once a token budget is set, it takes over as the primary bound for search results (this setting's own value is then unused there, though it still governs the keyword-search fallback, and the count served if token pricing itself fails). Does NOT raise the tool's hard ceiling of 10 passages per call - a model that reads max_results up to 10 from its own tool schema won't ask for more than that regardless of this setting.",
10214
+ description: "Passages the search_knowledge_base tool returns when a model call omits max_results, which is most calls. This is the exact bound while kbSearchResultTokenBudget is unset (0). Once a token budget is set, it takes over as the primary bound for search results (this setting's own value is then unused there, though it still governs the keyword-search fallback, and the count served if token pricing itself fails). Does NOT raise the tool's hard ceiling of 10 passages per call - a model that reads max_results up to 10 from its own tool schema won't ask for more than that regardless of this setting. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else.",
9273
10215
  category: "AI",
9274
10216
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9275
10217
  order: 5,
@@ -9281,7 +10223,7 @@ const settingsMap = {
9281
10223
  defaultValue: 0,
9282
10224
  min: 0,
9283
10225
  max: 2e4,
9284
- description: "Approximate tokens of served passage TEXT (post-trim, post-clip - what the model actually receives, not the raw stored chunk) the search_knowledge_base tool may return in one call. Counted with a fixed tokenizer as a proxy, not billed against any specific model. Replaces a passage count as the primary bound once set, since it is invariant to chunk size - a lake chunked smaller no longer silently returns less material for the same setting. 0 (default) disables it: search_knowledge_base then serves exactly kbSearchDefaultResults passages, unchanged from before this setting existed. The FIRST matching passage is always returned even if it alone exceeds the budget - a search that found something never returns nothing.",
10226
+ description: "Approximate tokens of served passage TEXT (post-trim, post-clip - what the model actually receives, not the raw stored chunk) the search_knowledge_base tool may return in one call. Counted with a fixed tokenizer as a proxy, not billed against any specific model. Replaces a passage count as the primary bound once set, since it is invariant to chunk size - a lake chunked smaller no longer silently returns less material for the same setting. 0 (default) disables it: search_knowledge_base then serves exactly kbSearchDefaultResults passages, unchanged from before this setting existed. The FIRST matching passage is always returned even if it alone exceeds the budget - a search that found something never returns nothing. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else.",
9285
10227
  category: "AI",
9286
10228
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9287
10229
  order: 6,
@@ -9293,12 +10235,50 @@ const settingsMap = {
9293
10235
  defaultValue: 0,
9294
10236
  min: 0,
9295
10237
  max: 100,
9296
- description: "Minimum cosine relevance, as a percent, a passage must clear to be returned by search_knowledge_base. 0 (default) matches current behavior (no relevance floor beyond a non-negative cosine score). Raising it lets breadth adapt per query - a narrow question can return fewer, more relevant passages instead of always padding out to the configured count. Cosine similarity is not comparable across embedding models: a floor tuned for one model can filter out an entire alternate model, when a lake mixes embedding models, more aggressively than intended. Start low and raise gradually while watching the tool's own retrieval-skipped notices.",
10238
+ description: "Minimum cosine relevance, as a percent, a passage must clear to be returned by search_knowledge_base. 0 (default) matches current behavior (no relevance floor beyond a non-negative cosine score). Raising it lets breadth adapt per query - a narrow question can return fewer, more relevant passages instead of always padding out to the configured count. Cosine similarity is not comparable across embedding models: a floor tuned for one model can filter out an entire alternate model, when a lake mixes embedding models, more aggressively than intended. Start low and raise gradually while watching the tool's own retrieval-skipped notices. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else.",
9297
10239
  category: "AI",
9298
10240
  group: API_SERVICE_GROUPS.EMBEDDING.id,
9299
10241
  order: 7,
9300
10242
  scope: { settableAt: ["organization", "owner"] }
9301
10243
  }),
10244
+ lakeMemoryRecallK: makeNumberSetting({
10245
+ key: "lakeMemoryRecallK",
10246
+ name: "Lake Memory Belief Budget",
10247
+ defaultValue: 24,
10248
+ min: 1,
10249
+ max: 200,
10250
+ int: true,
10251
+ description: "Most beliefs the lake memory hot-card injects on a Data-Lake-mode turn, shared across every lake in scope. Recall still applies its cosine floor and the source-reachability gate first, so raising this does not admit low-quality beliefs - it raises the ceiling on how many QUALIFYING beliefs can actually be used, which was pinned at 8 (inherited from personal-memento recall) on no evidence beyond that inheritance. The sibling lever on the same turn is Forced Retrieval Char Budget, which governs raw chunk text rather than extracted beliefs. Platform-only for now, unlike that sibling: this read goes through plain getSettingsValue, which ignores settableAt, so a scope block here would be silently inert - every override written against it would resolve to nothing. Pointing the read at the scoped resolver is the prerequisite, not extra metadata.",
10252
+ category: "AI",
10253
+ group: API_SERVICE_GROUPS.EMBEDDING.id,
10254
+ order: 8
10255
+ }),
10256
+ forcedRetrievalRelativeFloorPct: makeNumberSetting({
10257
+ key: "forcedRetrievalRelativeFloorPct",
10258
+ name: "Forced Retrieval Relative Floor (%)",
10259
+ defaultValue: 85,
10260
+ min: 0,
10261
+ max: 100,
10262
+ int: true,
10263
+ description: "How close to the best-scoring passage of the SAME turn a chunk must score to be injected on a Data-Lake-mode turn, as a percent of that top score. This is the floor that ranks; the absolute floor below only rejects. Unlike an absolute cosine line, it moves with the turn, so it keeps working when a corpus or an embedding model puts the whole score band somewhere else. Raising it injects fewer, more sharply-ranked passages and leaves char budget unspent; lowering it admits more of the tail. 0 disables the relative floor and leaves the absolute one as the only gate (the pre-#2497 behavior). The default is behavior-preserving rather than tuned: it admits everything the absolute floor admitted on the measured band, so it changes nothing until raised. Tune it AFTER an embedding-model change, never before - a migration shifts the band any value fitted to today would have been chosen against.",
10264
+ category: "AI",
10265
+ group: API_SERVICE_GROUPS.EMBEDDING.id,
10266
+ order: 9,
10267
+ scope: { settableAt: ["organization", "owner"] }
10268
+ }),
10269
+ forcedRetrievalMinSimilarityPct: makeNumberSetting({
10270
+ key: "forcedRetrievalMinSimilarityPct",
10271
+ name: "Forced Retrieval Absolute Floor (%)",
10272
+ defaultValue: 75,
10273
+ min: 1,
10274
+ max: 100,
10275
+ int: true,
10276
+ description: `Absolute minimum cosine similarity, as a percent, a chunk must clear to be injected on a Data-Lake-mode turn. This is a sanity floor for genuinely unrelated content, NOT the ranking gate - the relative floor above does the ranking. LEAVE IT AT 75 UNLESS YOU HAVE MEASURED YOUR OWN CORPUS: a raw cosine means nothing outside the embedding model it was fitted to, so while this reads 75 the server ignores it and applies the floor measured for whichever model your documents are actually embedded with (${forcedRetrievalFloorsBySpaceSummary}, and no absolute floor at all for a model nobody has measured - the relative floor still applies). Set any other value and the server uses exactly that, in every space, which is yours to get right: 75 against text-embedding-3-small sits above that band entirely and returns nothing on every query. Where this floor lands inside your band decides a lot - on one measured corpus 74 / 75 / 76 swung recall 91% / 65% / 40% - and the same 75 that is a cliff on one lake rejects nothing at all on another. Re-measure after changing the embedding model; the sweep tool is packages/scripts/retrieval/forcedFloorSweep.ts.`,
10277
+ category: "AI",
10278
+ group: API_SERVICE_GROUPS.EMBEDDING.id,
10279
+ order: 10,
10280
+ scope: { settableAt: ["organization", "owner"] }
10281
+ }),
9302
10282
  LakeAccessAuditRetentionDays: makeNumberSetting({
9303
10283
  key: "LakeAccessAuditRetentionDays",
9304
10284
  name: "Lake Access Audit Retention (days)",
@@ -9334,6 +10314,7 @@ const settingsMap = {
9334
10314
  }),
9335
10315
  dataLakeEmbeddingSpendEnabled: makeBooleanSetting({
9336
10316
  key: "dataLakeEmbeddingSpendEnabled",
10317
+ userReadable: true,
9337
10318
  name: "Data Lake Embedding Spend Enabled",
9338
10319
  defaultValue: true,
9339
10320
  description: "Master switch for data-lake embedding spend (ingestion, reprocessing, convergence). Off halts all provider embedding calls on those paths; cached embeddings still apply.",
@@ -9343,6 +10324,7 @@ const settingsMap = {
9343
10324
  }),
9344
10325
  dataLakeEmbeddingBudgetPerRunUsd: makeNumberSetting({
9345
10326
  key: "dataLakeEmbeddingBudgetPerRunUsd",
10327
+ userReadable: true,
9346
10328
  name: "Embedding Budget Per Run (USD)",
9347
10329
  defaultValue: 5,
9348
10330
  min: 0,
@@ -9396,6 +10378,17 @@ const settingsMap = {
9396
10378
  group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
9397
10379
  order: 6
9398
10380
  }),
10381
+ dataLakeEmbeddingMaxTokensPerMinute: makeNumberSetting({
10382
+ key: "dataLakeEmbeddingMaxTokensPerMinute",
10383
+ name: "Embedding Max Tokens Per Minute",
10384
+ defaultValue: DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_DEFAULT,
10385
+ min: 0,
10386
+ max: DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_MAX,
10387
+ description: "Most provider embedding TOKENS per minute across all data-lake work, which is the quantity providers actually meter. The calls-per-minute lever alone does not bound this: one call carries a whole batch of passages. Set it from your provider dashboard TPM, leaving headroom for query-side embedding (exempt, so a search never queues behind a backfill). 0 stops all calls.",
10388
+ category: "AI",
10389
+ group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
10390
+ order: 7
10391
+ }),
9399
10392
  dataLakeVectorizeChunkBatchSize: makeNumberSetting({
9400
10393
  key: "dataLakeVectorizeChunkBatchSize",
9401
10394
  name: "Vectorize Chunk Batch Size",
@@ -9405,7 +10398,7 @@ const settingsMap = {
9405
10398
  description: "How many chunks the chunk handler packs into one vectorize-queue message. Smaller batches smooth the fan-out; not a spend value, so min 1.",
9406
10399
  category: "AI",
9407
10400
  group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
9408
- order: 7
10401
+ order: 8
9409
10402
  }),
9410
10403
  dataLakeEmbeddingTierMultiplierIndividual: makeNumberSetting({
9411
10404
  key: "dataLakeEmbeddingTierMultiplierIndividual",
@@ -9416,7 +10409,7 @@ const settingsMap = {
9416
10409
  description: "Scales the per-run and per-lake embedding budgets for lakes owned by an individual user. 1 means those lakes get exactly the configured budgets; 0 stops them spending at all. The effective budget is still capped by the same hard rail as the untiered value.",
9417
10410
  category: "AI",
9418
10411
  group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
9419
- order: 8
10412
+ order: 9
9420
10413
  }),
9421
10414
  dataLakeEmbeddingTierMultiplierOrganization: makeNumberSetting({
9422
10415
  key: "dataLakeEmbeddingTierMultiplierOrganization",
@@ -9427,7 +10420,7 @@ const settingsMap = {
9427
10420
  description: "Scales the per-run and per-lake embedding budgets for lakes owned by an organization, which serve a whole team rather than one person. 0 stops org-owned lakes spending at all. The effective budget is still capped by the same hard rail as the untiered value.",
9428
10421
  category: "AI",
9429
10422
  group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
9430
- order: 9
10423
+ order: 10
9431
10424
  }),
9432
10425
  slackSigningSecret: makeStringSetting({
9433
10426
  key: "slackSigningSecret",
@@ -9529,6 +10522,7 @@ const settingsMap = {
9529
10522
  }),
9530
10523
  enableVoiceSession: makeBooleanSetting({
9531
10524
  key: "enableVoiceSession",
10525
+ userReadable: true,
9532
10526
  name: "Enable Voice Session",
9533
10527
  defaultValue: false,
9534
10528
  description: "Whether to enable the voice session.",
@@ -9538,6 +10532,7 @@ const settingsMap = {
9538
10532
  }),
9539
10533
  voiceV2Enabled: makeBooleanSetting({
9540
10534
  key: "voiceV2Enabled",
10535
+ userReadable: true,
9541
10536
  name: "Enable Voice v2 (Model-Agnostic)",
9542
10537
  defaultValue: false,
9543
10538
  description: "Gate for the Voice v2 feature (ElevenLabs Conversational AI + any B4M reasoning model). When disabled, /api/voice/v2/sessions returns 403.",
@@ -9557,6 +10552,7 @@ const settingsMap = {
9557
10552
  }),
9558
10553
  voiceSessionAiVoice: makeStringSetting({
9559
10554
  key: "voiceSessionAiVoice",
10555
+ userReadable: true,
9560
10556
  name: "Default Assistant Voice",
9561
10557
  defaultValue: "alloy",
9562
10558
  description: "The default voice for the assistant in the voice session.",
@@ -9858,16 +10854,6 @@ const settingsMap = {
9858
10854
  group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
9859
10855
  order: 80
9860
10856
  }),
9861
- EnableOptiHashiDefault: makeBooleanSetting({
9862
- key: "EnableOptiHashiDefault",
9863
- name: "OptiHashi: On by default for users",
9864
- defaultValue: false,
9865
- description: "When enabled, OptiHashi is active for users who have never explicitly toggled it.",
9866
- category: "Experimental",
9867
- group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
9868
- order: 81,
9869
- dependsOn: "EnableOptiHashi"
9870
- }),
9871
10857
  EnableLibreOncology: makeBooleanSetting({
9872
10858
  key: "EnableLibreOncology",
9873
10859
  name: "Enable LibreOncology",
@@ -10068,6 +11054,7 @@ const settingsMap = {
10068
11054
  }),
10069
11055
  orchestrationDefaults: makeObjectSetting({
10070
11056
  key: "orchestrationDefaults",
11057
+ userReadable: true,
10071
11058
  name: "Agent Orchestration Defaults",
10072
11059
  defaultValue: OrchestrationDefaultsSchema.parse({}),
10073
11060
  description: "Default ReAct profile for agentless executions (#8922). Drives allowed/denied tools, iteration ceilings, default thoroughness, and fallback models when the agent_executor is invoked without a persisted IAgent (e.g. the upcoming Agent-mode toggle).",
@@ -10138,6 +11125,15 @@ const settingsMap = {
10138
11125
  group: API_SERVICE_GROUPS.MODEL_DISCOVERY.id,
10139
11126
  order: 6
10140
11127
  }),
11128
+ modelDiscoveryProbeNewModels: makeBooleanSetting({
11129
+ key: "modelDiscoveryProbeNewModels",
11130
+ name: "Probe New Models for Dispatch",
11131
+ defaultValue: true,
11132
+ description: "Lets a write-mode run spend a forced one-tool call on a newly discovered OpenAI model to verify which token parameter and tool transport it takes, and turn its tools on. Off leaves every new OpenAI model with tools withheld until an operator writes the dispatch profile by hand.",
11133
+ category: "AI",
11134
+ group: API_SERVICE_GROUPS.MODEL_DISCOVERY.id,
11135
+ order: 7
11136
+ }),
10141
11137
  prReportRepo: makeStringSetting({
10142
11138
  key: "prReportRepo",
10143
11139
  name: "PR Report Repository",
@@ -10230,7 +11226,8 @@ const ModelTelemetrySchema = z$1.object({
10230
11226
  "google",
10231
11227
  "xai",
10232
11228
  "ollama",
10233
- "moonshot"
11229
+ "moonshot",
11230
+ "deepseek"
10234
11231
  ]),
10235
11232
  fallbackUsed: z$1.boolean(),
10236
11233
  fallbackReason: z$1.string().optional(),
@@ -10247,7 +11244,8 @@ const SystemPromptDetailSchema = z$1.object({
10247
11244
  "user",
10248
11245
  "project",
10249
11246
  "session",
10250
- "org"
11247
+ "org",
11248
+ "caller"
10251
11249
  ]),
10252
11250
  /** e.g., "date_context", "tool_guidance" */
10253
11251
  name: z$1.string(),
@@ -10641,6 +11639,7 @@ const PromptMetaContextSchema = z$1.object({
10641
11639
  }).optional(),
10642
11640
  lakeMemory: z$1.object({
10643
11641
  beliefCount: z$1.number(),
11642
+ beliefBudget: z$1.number().optional(),
10644
11643
  dataLakeTags: z$1.array(z$1.string())
10645
11644
  }).optional(),
10646
11645
  contextWindowUsage: z$1.object({
@@ -10685,7 +11684,15 @@ const PromptMetaPerformanceSchema = z$1.object({
10685
11684
  totalResponseTime: z$1.number().optional(),
10686
11685
  contextRetrievalTime: z$1.number().optional(),
10687
11686
  modelInferenceTime: z$1.number().optional(),
11687
+ /**
11688
+ * Time to First Visible Token: elapsed ms until the first chunk the user can actually
11689
+ * see. Left unset when a turn streamed nothing visible (thinking-only, or a turn that
11690
+ * errored before answering), so absence reads as "never rendered" rather than as fast.
11691
+ * Pair with firstChunkTime to tell a slow model from a long hidden-reasoning window.
11692
+ */
10688
11693
  firstTokenTime: z$1.number().optional(),
11694
+ /** Elapsed ms until the first chunk of any kind, including a hidden thinking block. */
11695
+ firstChunkTime: z$1.number().optional(),
10689
11696
  clientFirstTokenTime: z$1.number().optional(),
10690
11697
  streamingPerformance: z$1.object({
10691
11698
  chunkCount: z$1.number().optional(),
@@ -10766,11 +11773,16 @@ const CitableSourceSchema = z$1.object({
10766
11773
  * zero-result retrieval, so a turn that legitimately found nothing is indistinguishable from one
10767
11774
  * where retrieval never ran at all.
10768
11775
  *
10769
- * Deliberately holds NO counts and NO chunk/document identifiers. Counts already exist and are
10770
- * more precise: `citables.filter(c => c.type === 'document')` is deduped by id/url/title in
11776
+ * Holds NO chunk/document identifiers, and no DOCUMENT count. A document count already exists and
11777
+ * is more precise: `citables.filter(c => c.type === 'document')` is deduped by id/url/title in
10771
11778
  * `applyQuestStatusChanges`, while this shape cannot dedupe (no identifiers to dedupe by) and
10772
- * would have to sum - producing a second, disagreeing number for the same question. Similarity
10773
- * scores live on `LakeAccessEvent`, not here.
11779
+ * would have to sum - producing a second, disagreeing number for the same question.
11780
+ *
11781
+ * `injected` below is NOT that number and does not reopen it: it counts PASSAGES and characters,
11782
+ * neither of which `citables` can express - one document contributes many passages, and a turn
11783
+ * that injected nothing emits no citable to count at all. Similarity scores otherwise live on
11784
+ * `LakeAccessEvent`; the single `injected.topScore` is here because that row is written only on a
11785
+ * turn that grounded, so it cannot carry the near-miss score of a turn that grounded on nothing.
10774
11786
  *
10775
11787
  * CAUTION, not a guarantee: the absence of chunk/document identifiers is what keeps this shape
10776
11788
  * OUT of `promptMetaRedaction.ts`'s scope (that helper is a functionCalls-only denylist and would
@@ -10789,28 +11801,351 @@ const CitableSourceSchema = z$1.object({
10789
11801
  *
10790
11802
  * Absent-or-fully-present, matching `lakeMemory` above - see the Mongoose-side subSchema comment
10791
11803
  * in QuestModel.ts for why partial-write and default-array shapes are unsafe here.
11804
+ *
11805
+ * WIDER THAN ITS NAME SUGGESTS as of `mode` (#1394). The field is no longer written only when
11806
+ * retrieval ran: it is now seeded on every turn that could have retrieved (forced retrieval
11807
+ * enabled, or the knowledge tool offered), so `attempted: false` is a recorded fact rather than
11808
+ * an absence to be inferred. Presence therefore means "this turn was in a position to retrieve",
11809
+ * and turns with no knowledge in scope still carry no field at all. The distinction matters to a
11810
+ * rollup: absence is now ambiguous between "not a retrieval turn" and "written before this
11811
+ * existed", which is why `mode` documents its own date-bounding requirement.
10792
11812
  */
10793
11813
  const RetrievalSummarySchema = z$1.object({
10794
11814
  /** True once a retrieval-capable surface actually ran (not merely offered) this turn. */
10795
11815
  attempted: z$1.boolean(),
10796
11816
  /**
11817
+ * Present if and only if `attempted` is true - an outcome describes a run, and a turn that
11818
+ * never ran retrieval has none. A reader testing for a specific value is unaffected (absence
11819
+ * is not any of them); a reader switching exhaustively must handle undefined.
11820
+ *
10797
11821
  * 'ok' - ran, whether or not anything came back (the zero case is a legitimate 'ok').
10798
11822
  * 'no_lakes' - ran but the user had no entitled/selected lake in scope.
10799
- * 'failed' - threw; recall did not complete.
10800
- * On multiple retrieval calls within one turn, merge priority is failed > ok > no_lakes (see
10801
- * retrievalSummaryMerge.ts's mergeRetrievalSummary): a single failure is never masked by a later
10802
- * success or abstain, and a real success on one surface is never masked by another surface's
10803
- * "no lakes in scope" abstain in the same turn.
11823
+ * 'not_indexed' - ran to completion having compared nothing: the corpus in scope carries no
11824
+ * usable vector (never indexed, or embedded with a foreign model), so no passage was ever
11825
+ * scored against the query. Distinct from 'ok' because the library was not searched at all,
11826
+ * and reporting that as a topical zero ("your documents do not cover this") is exactly the
11827
+ * confident-wrong-answer this field exists to catch. Distinct from 'failed' because nothing
11828
+ * broke: the remedy is re-vectorizing, which the corpus owner can do themselves, and a retry
11829
+ * never helps.
11830
+ * COVERAGE: recorded by forced retrieval (KnowledgeRetrievalFeature's `scoredCount === 0`
11831
+ * exit) and by knowledgeBaseSearch, whose semantic arms carry the same verdict through to the
11832
+ * keyword arm's write - the two agree on "not one passage was compared against the query",
11833
+ * not on any withholding flag, so a relevance floor that emptied a real search and a partial
11834
+ * withholding alongside a real search both stay 'ok' on both surfaces.
11835
+ * knowledgeBaseRetrieve cannot reach this state: it fetches named files rather than ranking
11836
+ * against a query embedding, so it has no comparison to come up empty.
11837
+ * 'failed' - recall did not complete: it threw, OR the retrieval repository is not wired on
11838
+ * this host (the guards in ChatCompletionFeatures / knowledgeBaseSearch / knowledgeBaseRetrieve
11839
+ * record it without anything throwing). What separates it from 'not_indexed' is the remedy,
11840
+ * not the tempo: fix the outage or the host wiring, never re-index content. An unwired host
11841
+ * reports continuously too, so "chronic" alone does not pick out 'not_indexed'.
11842
+ * NOT this: a model-supplied argument that is not a well-formed id. knowledgeBaseRetrieve
11843
+ * shape-checks `file_id` and answers a malformed one as a single-file miss ('ok'), because the
11844
+ * remedy is for the model to search for the right id - there is nothing for an operator to
11845
+ * fix. It is logged rather than counted here, so the rate stays observable without this field
11846
+ * reporting an outage that is not happening.
11847
+ * On multiple retrieval calls within one turn, merge priority is failed > not_indexed > ok >
11848
+ * no_lakes (see retrievalSummaryMerge.ts's mergeRetrievalSummary): a single failure is never
11849
+ * masked by a later success or abstain, an unsearchable corpus outranks a legitimate zero so a
11850
+ * success on another surface cannot erase it, and a real success is never masked by another
11851
+ * surface's "no lakes in scope" abstain in the same turn.
10804
11852
  */
10805
11853
  outcome: z$1.enum([
10806
11854
  "ok",
10807
11855
  "no_lakes",
11856
+ "not_indexed",
10808
11857
  "failed"
10809
- ]),
11858
+ ]).optional(),
11859
+ /**
11860
+ * Whether forced retrieval was ENABLED for this turn, independent of whether it then ran.
11861
+ *
11862
+ * This is what makes the optional path measurable. `attempted` says retrieval happened;
11863
+ * without `mode` there is no way to ask the complementary question - of the turns where the
11864
+ * model was merely OFFERED the knowledge tools, how often did it choose to retrieve - because
11865
+ * a forced turn and an optional turn both land as `attempted: true`.
11866
+ *
11867
+ * Optional on the schema, and absence NEVER means 'optional' - it means unclassified. Two
11868
+ * sources, one historical and one ongoing: turns recorded before this field landed, and
11869
+ * agent-mode runs, which write a retrieval summary through `persistRunAsQuest` but never pass
11870
+ * the seed site at the `offeredTools` write in ChatCompletionProcess. So date-bounding a rollup
11871
+ * removes the first source but not the second; count the unclassified bucket rather than
11872
+ * assuming it empties.
11873
+ */
11874
+ mode: z$1.enum(["forced", "optional"]).optional(),
11875
+ /**
11876
+ * Whether the knowledge-base when-to-retrieve guidance section actually shipped in this turn's
11877
+ * tool prompt. Written only on turns that were OFFERED the knowledge tool, so absence means
11878
+ * "not an offered turn" or "recorded before this field landed" - it never means "cleared".
11879
+ *
11880
+ * `false` is the load-bearing value here, not filler. Clearing the KnowledgeBaseRetrievalPrompt
11881
+ * setting is the section's only off switch, so a turn recording `false` is the CONTROL arm of
11882
+ * the A/B this field exists to make readable. Anything merging or folding this must preserve an
11883
+ * explicit `false` rather than collapse it into absent - see mergeRetrievalSummary, which uses
11884
+ * `??` and deliberately not `||` for that reason.
11885
+ *
11886
+ * MUST STAY IN SYNC with TWO gates, not one. ToolBuilder.buildToolPrompt emits the section iff
11887
+ * the tool is offered AND the guidance string is non-empty; filterByPromptMode then drops the
11888
+ * whole `toolPrompt` source, which no promptMode admits, so an offered tool is not sufficient.
11889
+ * The ChatCompletionProcess seed site conjoins all three, and hands the first two to
11890
+ * buildToolPrompt as the same consts, so the flag and the actual emission cannot drift.
11891
+ */
11892
+ knowledgeBaseGuidanceInjected: z$1.boolean().optional(),
11893
+ /**
11894
+ * Why the forced arm did not run on a turn that had it enabled. Only ever set with
11895
+ * `mode: 'forced'`, and only for the deliberate suppressions in
11896
+ * ChatCompletionFeatures.getContextMessages - a forced turn that ran and failed reports that
11897
+ * through `outcome`, not here.
11898
+ *
11899
+ * These turns are the reason this field exists: forced retrieval is configured, a rule
11900
+ * suppresses it, and the model falls back to the offered tool. That is exactly the population
11901
+ * the per-turn routing question is about, and before this it was indistinguishable from a turn
11902
+ * where forced retrieval was never configured at all.
11903
+ */
11904
+ forcedSkipReason: z$1.enum(["attached_files", "personal_corpus"]).optional(),
10810
11905
  /** Which retrieval-capable surface(s) ran this turn, e.g. 'lake-memory', 'knowledgeBaseSearch'. */
10811
11906
  surfaces: z$1.array(z$1.string()),
10812
11907
  /** Lakes resolved at the moment retrieval ran, stamped point-in-time (not read live from the session). */
10813
- dataLakeTags: z$1.array(z$1.string())
11908
+ dataLakeTags: z$1.array(z$1.string()),
11909
+ /**
11910
+ * The lake scope the turn's retrieval surfaces WOULD have searched, resolved at the seed site
11911
+ * whether or not any of them ran: the caller's accessible lakes narrowed to the session
11912
+ * (narrowLakeAccessToSession), or empty where the corpus is personal and the lake arms are
11913
+ * suppressed. `dataLakeTags` is the other half of the pair and answers a different question -
11914
+ * which lakes retrieval ACTUALLY used - so on a turn where retrieval never ran that one is empty
11915
+ * while this one still names whatever was in scope.
11916
+ *
11917
+ * EXISTS FOR THE OFFLINE REPLAY. `answerability` is reconstructed after the fact, and without a
11918
+ * recorded scope the replay had to rebuild one from the session's `retrievalTags` as they stand
11919
+ * at replay time - a session whose lake selection had since changed was replayed against a
11920
+ * corpus its turn never had, with nothing to flag it. Recording it here removes that drift for
11921
+ * every turn seeded after this landed; `probedAt` still discloses the content drift, which no
11922
+ * amount of recording can fix.
11923
+ *
11924
+ * Absence means NOT RECORDED (a turn predating this, or one whose `retrieval` was written only
11925
+ * by a surface rather than by the seed) - never "no lakes in scope", which is present-and-empty.
11926
+ * The replay must keep those apart: probing an unrecorded turn would mean inventing a scope,
11927
+ * which is the approximation this field exists to end.
11928
+ *
11929
+ * Only the seed writes it, so mergeRetrievalSummary carries it first-writer-wins rather than
11930
+ * unioning: a later surface write asserting a narrower scope must not be able to widen the
11931
+ * recorded one, and a union across the two would mean neither.
11932
+ */
11933
+ lakeScope: z$1.array(z$1.string()).optional(),
11934
+ /**
11935
+ * Ids of the lakes whose `systemPrompt` was injected this turn (getAccessibleDataLakePrompts),
11936
+ * across every injection site (forced retrieval and the model-driven knowledge tools). NOT the
11937
+ * prompt text itself - that already reaches the model in the completion, and copying it here
11938
+ * widens exposure for nothing. Absent means no injection site ran; present-and-empty means one
11939
+ * ran but nothing qualified (untrusted, or an empty systemPrompt).
11940
+ */
11941
+ injectedLakePromptIds: z$1.array(z$1.string()).optional(),
11942
+ /** mementoCount/mementoIds precedent: mirrors injectedLakePromptIds.length. */
11943
+ injectedLakePromptCount: z$1.number().optional(),
11944
+ /**
11945
+ * How much retrieved content actually reached the model this turn: `chunks` passages totalling
11946
+ * `chars` characters of retrieved CONTENT (headings and framing excluded, so the number means
11947
+ * the same thing on every surface), plus `topScore`, the best similarity among the compared
11948
+ * passages the reporting surface can SEE - which is not the same population on every surface.
11949
+ * Forced retrieval scores every chunk itself and so reports true near-misses; knowledgeBaseSearch's
11950
+ * semantic arm only ever sees `minScore` survivors, and reports no `topScore` at all on a starve,
11951
+ * so a sub-floor near-miss there is invisible rather than recorded.
11952
+ *
11953
+ * PRESENCE CONTRACT: present if and only if at least one surface COMPLETED a search this turn.
11954
+ * `chunks: 0` is a RECORDED STARVE - the library was searched and nothing was injected, which is
11955
+ * the case this field exists to make visible: without it, a forced-retrieval turn that injected
11956
+ * nothing is byte-identical to one that injected its whole character budget (both `outcome:
11957
+ * 'ok'`). Absence means the volume is UNKNOWN, which is what a turn carries when no surface
11958
+ * completed a search: retrieval was never attempted, nothing was in scope to search
11959
+ * ('no_lakes'), or the one surface that ran broke mid-flight, where a zero would be a lie.
11960
+ * A surface that completed but CANNOT know the turn's passage volume also stays silent rather
11961
+ * than claiming a zero - knowledgeBaseSearch's keyword arm on a hit is the case: it injects
11962
+ * file metadata and hands the model retrieve_knowledge_content, which injects the text and
11963
+ * reports no volume, so its zero would survive the merge as a starve that did not happen.
11964
+ *
11965
+ * KNOWN HOLE in that rule, while retrieve_knowledge_content stays uninstrumented: a recorded zero
11966
+ * is not PROOF of a starve. Forced retrieval and the knowledge tools are not mutually exclusive
11967
+ * (ChatCompletionProcess seeds on `forcedRetrievalEnabled || knowledgeToolOffered`), so the forced
11968
+ * arm can complete empty, write its honest zero, and the model can then ground the same turn
11969
+ * through retrieve_knowledge_content, which contributes no volume to oppose it. The zero is
11970
+ * per-surface-truthful and turn-level-misleading. Any rollup counting starves should treat a zero
11971
+ * as "nothing was injected by a surface that reports volume" and, until that tool reports its own,
11972
+ * cross-check `functionCalls` before calling the turn ungrounded.
11973
+ *
11974
+ * Per SURFACE, not per turn: a surface that breaks contributes nothing while a surface that
11975
+ * completed alongside it still reports its own volume, so a turn CAN read 'failed' next to a
11976
+ * recorded zero. That pairing means "one surface broke, and everything that did finish injected
11977
+ * nothing" - which is exactly what a reader needs, and strictly more than the outcome alone.
11978
+ *
11979
+ * SUMMED across surfaces, so this field and `outcome` can legitimately disagree in tone on a
11980
+ * multi-surface turn: forced retrieval grounding on 12 passages while knowledgeBaseSearch throws
11981
+ * gives `outcome: 'failed'` alongside `chunks: 12`. That is correct - `outcome` is worst-of,
11982
+ * `injected` is sum-of-completions.
11983
+ *
11984
+ * `topScore` is optional because only cosine-similarity surfaces have one to report. Lake
11985
+ * memory's belief `relevance` is a different scale and forced retrieval's pre-scan value is a
11986
+ * -1 sentinel; neither is ever written here, because a `max` across mixed scales, or against a
11987
+ * sentinel, is a number that reads as a similarity and is not one.
11988
+ *
11989
+ * Date-bound any rollup, the same caveat `mode` documents on itself: turns recorded before this
11990
+ * landed carry no volume, and no backfill is possible - the volume of a past turn is gone.
11991
+ *
11992
+ * `preRelativeFloorCandidates` and `postRelativeFloorCandidates` are the ONE pair here that is
11993
+ * not "what reached the model": `ranked.length` and `scored.length` in KnowledgeRetrievalFeature
11994
+ * - the candidates left after the absolute similarity floor, and after the relative floor
11995
+ * trims them. `chunks` is what survived the char budget on top of that, so the three
11996
+ * numbers bracket two independent trimmers:
11997
+ *
11998
+ * pre -> [relative floor] -> post -> [char budget] -> chunks
11999
+ *
12000
+ * They exist so a low `chunks` is diagnosable - a small corpus and a floor that trimmed a large
12001
+ * pool end in the same `chunks`. `pre - post` is the floor's own effect and nothing else;
12002
+ * `pre - chunks` is NOT, because the budget trims the same walk. Both optional: only forced
12003
+ * retrieval computes a ranked pool, a surface without one (lake memory, the knowledge tools)
12004
+ * never writes either, and absence must not read as zero candidates. SUMMED like `chunks`, with
12005
+ * the same absent-is-not-zero handling as `topScore`.
12006
+ *
12007
+ * COMPARE THE PAIR ONLY TO ITSELF, never to `chunks`, unless `surfaces` is forced retrieval
12008
+ * alone. `chunks` and `chars` sum across ALL surfaces while this pair is forced-only, so a mixed
12009
+ * turn can store `chunks` above `pre` - inverting the relationship the pair exposes. Lake memory
12010
+ * is the common case, not the exotic one: it is enabled inside the same forced-retrieval gate,
12011
+ * so on a lake-memory lake it writes on nearly every forced turn. Its chunks can be backed out
12012
+ * via `context.lakeMemory.beliefCount` (approximately - that count is pre-sanitization); its
12013
+ * CHARS land only inside the shared sum, with no per-surface field to subtract them back out, so
12014
+ * `chars` cannot be decontaminated at all. `pre - post` needs neither, which is the point of
12015
+ * storing both.
12016
+ *
12017
+ * BOTH SATURATE, so `pre` counts what the SCAN REACHED, not what the corpus holds: `pool` is
12018
+ * truncated in-scan at FORCED_RETRIEVAL_MAX_SCORED_CHUNKS (256), over a scan itself bounded by
12019
+ * FORCED_RETRIEVAL_MAX_SCANNED_CHUNKS (4000) across FORCED_RETRIEVAL_MAX_CANDIDATE_FILES (100).
12020
+ * 2000 qualifying chunks and 300 both record 256; above the cap a rollup is a plateau.
12021
+ */
12022
+ injected: z$1.object({
12023
+ chunks: z$1.number(),
12024
+ chars: z$1.number(),
12025
+ topScore: z$1.number().optional(),
12026
+ preRelativeFloorCandidates: z$1.number().optional(),
12027
+ postRelativeFloorCandidates: z$1.number().optional()
12028
+ }).optional(),
12029
+ /**
12030
+ * Could the corpus in scope have answered this turn, whether or not the model went looking?
12031
+ *
12032
+ * The denominator the optional-path retrieval rate has always been missing (#1394). A rate of
12033
+ * "the model retrieved on 20% of offered turns" cannot say whether the other 80% were misses or
12034
+ * turns with nothing to find, and the two argue for opposite things: the first for routing work,
12035
+ * the second for leaving the optional path alone. Crossing this field with the rate separates
12036
+ * them.
12037
+ *
12038
+ * THE ONLY FIELD IN THIS BLOCK NOT WRITTEN BY THE TURN. Every sibling is stamped point-in-time
12039
+ * while the turn runs; this one is written afterwards by an offline replay
12040
+ * (packages/scripts/retrieval/answerability-replay.ts) that re-scores the recorded prompt against
12041
+ * the corpus. That is deliberate - the population it exists to measure is the turns where
12042
+ * retrieval did NOT run, so computing it live would mean adding a full brute-force chunk scan
12043
+ * (ChatCompletionFeatures' forced path, which has no ANN index) to exactly the turns that pay
12044
+ * nothing for retrieval today. The measurement is not worth that latency on live traffic.
12045
+ *
12046
+ * BEING A RECONSTRUCTION, IT CARRIES DRIFTS THE OTHER FIELDS DO NOT:
12047
+ * 1. Corpus CONTENT moves. A document added or reindexed between the turn and the replay is
12048
+ * scored as though it had been there. `probedAt` discloses the gap; a replay run long after
12049
+ * the window is weak evidence, not strong.
12050
+ * 2. Corpus SCOPE no longer drifts: the seed records the turn's resolved scope in `lakeScope`
12051
+ * and the replay probes that, so a session whose lake selection has since changed is still
12052
+ * scored against the lakes its turn actually had. The cost is coverage rather than accuracy -
12053
+ * a turn with no recorded scope is skipped instead of approximated, so every turn predating
12054
+ * the field is outside the measurement.
12055
+ * 3. The QUESTION can move out from under it. The probe is keyed to the quest, not to the
12056
+ * prompt text it scored, so a turn whose prompt is later rewritten in place keeps a probe
12057
+ * describing the question it used to ask. mergeRetrievalSummary preserves the probe across
12058
+ * a runtime write deliberately - dropping it would erase the backfill - so nothing
12059
+ * invalidates a stale one. Re-run the replay with --force over a window whose turns were
12060
+ * edited.
12061
+ *
12062
+ * RAW SCORE, NOT A VERDICT, so the cutoff lives in the reader. summarizeOptionalPathRetrieval
12063
+ * applies it at fold time, which lets the same replay be re-thresholded without re-running -
12064
+ * the point of storing the number, given the two live floors disagree by construction (forced
12065
+ * retrieval's absolute default is 0.75, the knowledge tool's is 0).
12066
+ *
12067
+ * `topScore` is the same raw cosine scale as `injected.topScore` and comparable to it. It is NOT
12068
+ * comparable to lake memory's belief relevance, for the reason `injected` documents at length.
12069
+ *
12070
+ * `scanTruncated` inherits forced retrieval's saturation: the replay bounds its scan the same
12071
+ * way, so a low `topScore` on a truncated scan is not proof the corpus lacked an answer - it is
12072
+ * proof the part that was scanned did. Treat those turns as unknown rather than as negatives.
12073
+ *
12074
+ * Absence means NOT PROBED - never "not answerable". Every turn predating the replay, and every
12075
+ * turn the replay skipped or failed on, is absent, so a fold must keep it as its own arm rather
12076
+ * than letting it fall in with the negatives.
12077
+ */
12078
+ answerability: z$1.object({
12079
+ /** Best cosine the replay found across the reconstructed corpus. */
12080
+ topScore: z$1.number(),
12081
+ /** Chunks at or above `floor`. Separates "one lucky match" from "a rich seam". */
12082
+ candidatesAboveFloor: z$1.number(),
12083
+ /** The absolute floor the replay counted `candidatesAboveFloor` against, as a fraction. */
12084
+ floor: z$1.number(),
12085
+ /** The scan hit its chunk ceiling, so `topScore` is a floor on the true best, not the best. */
12086
+ scanTruncated: z$1.boolean(),
12087
+ /** When the replay ran, NOT when the turn ran - the disclosure for content drift above. */
12088
+ probedAt: JsonSafeDate
12089
+ }).optional(),
12090
+ /**
12091
+ * Which of this turn's injected lake prompt ids were BOTH in the session's pre-authorized (manage-
12092
+ * but-not-member admission) set AND injected on this turn - see unionPreauthorizedLakeAccess and
12093
+ * pages/api/sessions/create.ts. A subset of injectedLakePromptIds, never a superset. Narrows the
12094
+ * session's static `preauthorizedLakeIds` (what was ADMITTED) to what a given turn actually used.
12095
+ *
12096
+ * MEMBERSHIP, NOT CAUSATION. An admitted lake the caller could already reach - its creator, or a
12097
+ * member of its org - injects through the ordinary trust arm and is listed here all the same, so a
12098
+ * non-empty value does not prove the admission is what made the injection possible. Absent means no
12099
+ * admitted id was among this turn's injections, including every turn on a session with none.
12100
+ */
12101
+ preauthorizedLakeIdsUsed: z$1.array(z$1.string()).optional(),
12102
+ /**
12103
+ * Which of this turn's injected lake prompt ids the caller holds an owner/curator GRANT on - the
12104
+ * per-arm sibling of `preauthorizedLakeIdsUsed`, and the reason both exist: `injectedLakePromptIds`
12105
+ * records THAT a lake's systemPrompt entered the turn, these two record WHICH ARM admitted it.
12106
+ * A subset of injectedLakePromptIds, never a superset. Derived at both injection sites via
12107
+ * grantedLakeIdsUsedFor.
12108
+ *
12109
+ * THE ARM WORTH NAMING SEPARATELY: the grant arm (#2495) is the only one that can cross an org
12110
+ * boundary - `grantLakeAccess` can hand a CURATOR grant to an arbitrary cross-tenant user, and
12111
+ * that grant carries injection trust. The creator and org arms cannot reach past one org, and a
12112
+ * reader grant is excluded permanently, so a lake listed here is the case an operator auditing
12113
+ * cross-tenant prompt influence is actually looking for.
12114
+ *
12115
+ * MEMBERSHIP, NOT CAUSATION, the same caveat the field above carries: a granted lake its holder
12116
+ * could already reach (they created it, or they are in its org) injects through the ordinary
12117
+ * trust arm and is listed here anyway, so a non-empty value does not prove the grant is what
12118
+ * made the injection possible. The two fields OVERLAP for that reason - a lake can appear in
12119
+ * both - so they are not a partition of injectedLakePromptIds and must not be counted as one.
12120
+ *
12121
+ * Absent means no injected id was grant-reached, including every turn where the grant read
12122
+ * FAILED (it is fail-quiet by design - telemetry must not drop the injection it records), so
12123
+ * absence is weaker evidence than presence. Date-bound any rollup: turns predating this field
12124
+ * carry nothing, and no backfill is possible - a past turn's grant rows have moved on.
12125
+ */
12126
+ grantedLakeIdsUsed: z$1.array(z$1.string()).optional()
12127
+ });
12128
+ /**
12129
+ * Why a grounded turn's library scan stopped short of the whole library.
12130
+ *
12131
+ * Written ONLY on a partially-covered turn (reportCoverage returns early otherwise), so presence
12132
+ * means "partial" and `partial` is always true - the flag is explicit anyway because a reader
12133
+ * checking `retrievalCoverage.partial` should not have to know that absence is the other half of
12134
+ * the contract.
12135
+ *
12136
+ * Single producer (ChatCompletionFeatures.reportCoverage), which is why - unlike `warnings`,
12137
+ * `citables` and `retrieval` - this field needs no merge case in applyQuestStatusChanges: a
12138
+ * later tool-arm write that omits it is preserved by the one-level spread.
12139
+ *
12140
+ * `reasons` is the same diagnostic prose the warnings entry interpolates. It is shown to the
12141
+ * reader behind a disclosure rather than in the banner body, because only some reasons are
12142
+ * actionable (a document mid-reindex returns on its own; a per-turn chunk budget does not).
12143
+ */
12144
+ const RetrievalCoverageSchema = z$1.object({
12145
+ /** Always true - see the presence contract above. */
12146
+ partial: z$1.boolean(),
12147
+ /** One entry per distinct cause, e.g. a candidate cap, a scan budget, an embedding mismatch. */
12148
+ reasons: z$1.array(z$1.string())
10814
12149
  });
10815
12150
  z$1.object({
10816
12151
  model: PromptMetaModelSchema.optional(),
@@ -10820,6 +12155,9 @@ z$1.object({
10820
12155
  * deliberately: applyQuestStatusChanges does a one-level spread merge, so a field nested under
10821
12156
  * `context` would be replaced wholesale by any tool-arm write instead of merging. */
10822
12157
  retrieval: RetrievalSummarySchema.optional(),
12158
+ /** Partial-grounding-coverage detail - see RetrievalCoverageSchema. Top-level for the same
12159
+ * one-level-spread-merge reason as `retrieval` above. */
12160
+ retrievalCoverage: RetrievalCoverageSchema.optional(),
10823
12161
  functionCalls: z$1.array(PromptMetaFunctionCallSchema).optional(),
10824
12162
  /**
10825
12163
  * Names of the tools actually offered to the model this turn - the output of `buildTools`
@@ -11513,6 +12851,7 @@ z.object({
11513
12851
  requiredUserTag: z.union([z.literal(""), z.string().min(1).max(100)]).optional(),
11514
12852
  requiredEntitlement: z.union([z.literal(""), z.string().min(3).max(100).refine((s) => s.includes(":") && s.split(":").every((part) => part.length > 0), "Entitlement key must be namespaced with non-empty parts (e.g. \"product:pro\")")]).optional(),
11515
12853
  auditQueryTextEnabled: z.boolean().optional(),
12854
+ lakeMemoryEnabled: z.boolean().optional(),
11516
12855
  requiredPassageTokenTarget: z.number().int().min(64).max(OVERSIZED_PASSAGE_TOKEN_THRESHOLD).nullable().optional()
11517
12856
  });
11518
12857
  z.object({
@@ -11587,6 +12926,7 @@ tags: z.array(TaxonomyTagInput).max(100) });
11587
12926
  z.object({
11588
12927
  /** User description of the data (helps the AI) */
11589
12928
  context: z.string().max(2e3).optional() });
12929
+ z.object({ tags: z.array(z.string().min(1).max(130).regex(/^[^\r\n]*$/)).max(100) });
11590
12930
  z.object({ hashes: z.array(z.string().regex(sha256Regex)).min(1).max(500) });
11591
12931
  const SyncDeltaFileEntry = z.object({
11592
12932
  relativePath: z.string(),
@@ -11618,55 +12958,6 @@ z.object({
11618
12958
  skip: z.array(z.string())
11619
12959
  })
11620
12960
  });
11621
- z$1.enum([
11622
- "inject",
11623
- "auto-fire",
11624
- "hidden"
11625
- ]);
11626
- /**
11627
- * Modes acceptable at AUTHORING time. 'hidden' is intentionally excluded until
11628
- * the host has true hidden-send support - accepting it would persist a value
11629
- * that silently behaves as 'auto-fire' (a surprising downgrade). It stays in
11630
- * ExecutionModeSchema/the stored enum for forward-compat.
11631
- */
11632
- const AuthorableExecutionModeSchema = z$1.enum(["inject", "auto-fire"]);
11633
- /**
11634
- * Tools a prompt may require - constrained to the host's closed tool set, MINUS
11635
- * integration-gated tools that act on the caller's own credentials/account. A
11636
- * shared system prompt must not be able to inject e.g. blog-publishing into a
11637
- * non-author's session via requiredTools. (Per-user entitlement of the remaining
11638
- * tools is still the chat pipeline's responsibility - see follow-up note in the
11639
- * briefcase blueprint; this allowlist is the storage-layer floor.)
11640
- */
11641
- const BRIEFCASE_DISALLOWED_TOOLS = [
11642
- "blog_publish",
11643
- "blog_edit",
11644
- "blog_draft"
11645
- ];
11646
- const BriefcaseRequiredToolsSchema = z$1.array(b4mLLMTools.refine((t) => !BRIEFCASE_DISALLOWED_TOOLS.includes(t), "This tool is not permitted in a briefcase prompt")).max(16);
11647
- z$1.string().regex(/^[a-f0-9]{24}$/i, "Invalid prompt id");
11648
- const PROMPT_TEXT_MAX = 16e3;
11649
- const TAGS_MAX = 20;
11650
- z$1.object({
11651
- type: z$1.string().min(1).max(100),
11652
- name: z$1.string().min(1).max(200),
11653
- description: z$1.string().max(500).optional(),
11654
- promptText: z$1.string().min(1).max(PROMPT_TEXT_MAX),
11655
- tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
11656
- executionMode: AuthorableExecutionModeSchema.optional(),
11657
- requiredTools: BriefcaseRequiredToolsSchema.optional()
11658
- }).partial();
11659
- /**
11660
- * One catalog sub-query. Exactly one selector is used, in precedence order:
11661
- * `personal` (resolved to the caller server-side) > `tags` > `type`.
11662
- */
11663
- const PromptBatchQuerySchema = z$1.object({
11664
- key: z$1.string().min(1).max(100),
11665
- tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
11666
- type: z$1.string().max(100).optional(),
11667
- personal: z$1.boolean().optional()
11668
- });
11669
- z$1.object({ queries: z$1.array(PromptBatchQuerySchema).min(1).max(32).refine((qs) => new Set(qs.map((q) => q.key)).size === qs.length, { message: "Batch query keys must be unique" }) });
11670
12961
  z$1.string().regex(/^[a-f0-9]{24}$/i, "Invalid template id");
11671
12962
  /**
11672
12963
  * The bound model. Reuses the legacy-remap preprocess so a template saved under
@@ -11969,6 +13260,31 @@ z$1.object({
11969
13260
  "grounded",
11970
13261
  "surface"
11971
13262
  ]).optional(),
13263
+ /**
13264
+ * Suppress OUR server-side auto-offers without entering a promptMode. Exists because promptMode
13265
+ * was the only switch for the offer and it also strips every authored prompt, so no caller could
13266
+ * have an arm that went unoffered AND kept the abstention licence.
13267
+ *
13268
+ * Gates the three auto-add sites (the knowledge offer in resolveEnabledTools, the navigate_view
13269
+ * auto-add, the blog/skill gate), unioned with `Boolean(promptMode)` by
13270
+ * resolveSkipAutoOffers. A force-on, not an override: `false` under a promptMode still suppresses.
13271
+ * Withholding navigate_view also drops the viewRegistry system block, which only describes it.
13272
+ *
13273
+ * Withholds the OFFER, not knowledge: `session.forceKnowledgeRetrieval` is untouched, and an
13274
+ * already-attached corpus is inlined rather than deferred to the tool. An arm that must see no
13275
+ * knowledge at all also needs a session with no attachments and forced retrieval off.
13276
+ */
13277
+ skipAutoOffers: z$1.boolean().optional(),
13278
+ /**
13279
+ * Caller-supplied system-prompt text. Rendered as a defended, deference-postured block
13280
+ * appended last in the system-prompt stack. Reached by both POST /api/chat and /api/ai/llm.
13281
+ *
13282
+ * This cap is the universal backstop, not a duplicate of a route check: the parse that opens
13283
+ * invoke() runs outside any try and before a quest row is written, so it holds for every caller
13284
+ * including ones that pass through no route schema. Do not drop it on the assumption that
13285
+ * whoever called validated first.
13286
+ */
13287
+ systemPrompt: z$1.string().max(PROMPT_TEXT_MAX).optional(),
11972
13288
  /** Whether Mementos is enabled */
11973
13289
  enableMementos: z$1.boolean().optional(),
11974
13290
  /** Whether Artifacts is enabled */
@@ -12942,7 +14258,7 @@ Array.from(new Set([
12942
14258
  id: "opti.root",
12943
14259
  section: "opti",
12944
14260
  label: "OptiHashi Home",
12945
- description: "The OptiHashi Optimizer landing page showing all 8 pattern family cards",
14261
+ description: "The OptiHashi Optimizer landing page showing the pattern family cards",
12946
14262
  navigationType: "route",
12947
14263
  target: "/opti",
12948
14264
  keywords: [
@@ -13843,6 +15159,66 @@ Array.from(new Set([
13843
15159
  const [, top] = v.target.split("/");
13844
15160
  return `/${top}`;
13845
15161
  })));
15162
+ function getHeader(headers, name) {
15163
+ if (!headers || typeof headers !== "object") return null;
15164
+ if (typeof headers.get === "function") {
15165
+ const value = headers.get(name);
15166
+ return typeof value === "string" ? value : null;
15167
+ }
15168
+ const value = headers[name] ?? headers[name.toLowerCase()];
15169
+ return typeof value === "string" ? value : null;
15170
+ }
15171
+ function parseCount(value) {
15172
+ if (value === null) return null;
15173
+ const trimmed = value.trim();
15174
+ if (!trimmed) return null;
15175
+ const parsed = Number(trimmed);
15176
+ return Number.isFinite(parsed) ? parsed : null;
15177
+ }
15178
+ const UNIT_MS = {
15179
+ ms: 1,
15180
+ s: 1e3,
15181
+ m: 6e4,
15182
+ h: 36e5
15183
+ };
15184
+ const DURATION_PART = /(\d+(?:\.\d+)?)(ms|h|m|s)/g;
15185
+ /**
15186
+ * Parse a Go-style duration ("6ms", "0s", "1m30s", "1h2m3s") to milliseconds.
15187
+ *
15188
+ * Exported for its own tests: it is the part of this module that can be wrong in a way the
15189
+ * numbers still look plausible.
15190
+ */
15191
+ function parseDurationMs(value) {
15192
+ if (typeof value !== "string") return null;
15193
+ const trimmed = value.trim();
15194
+ if (!trimmed) return null;
15195
+ DURATION_PART.lastIndex = 0;
15196
+ let total = 0;
15197
+ let matched = 0;
15198
+ let consumed = 0;
15199
+ for (const part of trimmed.matchAll(DURATION_PART)) {
15200
+ total += Number(part[1]) * UNIT_MS[part[2]];
15201
+ consumed += part[0].length;
15202
+ matched += 1;
15203
+ }
15204
+ if (matched === 0 || consumed !== trimmed.length) return null;
15205
+ return total;
15206
+ }
15207
+ /** Read both rate-limit dimensions off a provider response. */
15208
+ function parseEmbeddingRateLimitHeaders(headers) {
15209
+ return {
15210
+ limitTokens: parseCount(getHeader(headers, "x-ratelimit-limit-tokens")),
15211
+ limitRequests: parseCount(getHeader(headers, "x-ratelimit-limit-requests")),
15212
+ remainingTokens: parseCount(getHeader(headers, "x-ratelimit-remaining-tokens")),
15213
+ remainingRequests: parseCount(getHeader(headers, "x-ratelimit-remaining-requests")),
15214
+ resetTokensMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-tokens")),
15215
+ resetRequestsMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-requests"))
15216
+ };
15217
+ }
15218
+ /** True when the provider reported at least one usable ceiling. */
15219
+ function hasUsableLimits(snapshot) {
15220
+ return snapshot.limitTokens !== null || snapshot.limitRequests !== null;
15221
+ }
13846
15222
  dayjs.extend(utc);
13847
15223
  dayjs.extend(timezone);
13848
15224
  dayjs.extend(relativeTime);
@@ -13911,16 +15287,34 @@ function isUserInitiatedAbort(error, userSignal) {
13911
15287
  return isAbortError && !userSignal;
13912
15288
  }
13913
15289
  /**
13914
- * Extract retry delay from error response (e.g., Retry-After header)
15290
+ * A Retry-After hint is only useful if it asks us to wait. `Retry-After: 0`, a negative value, or an
15291
+ * HTTP date that has already passed all carry no timing information - and every `withRetry` in this
15292
+ * repo treats a non-null hint as authoritative *over* its exponential backoff, so returning 0 does
15293
+ * not mean "wait a moment", it means "abandon the backoff entirely and retry immediately".
15294
+ *
15295
+ * That inverts the retry budget exactly when it matters: a server sends Retry-After when it is
15296
+ * already struggling, so honouring a zero turns the remaining attempts into an instant burst against
15297
+ * a service asking for room. Null instead, so the caller falls through to its own backoff.
15298
+ *
15299
+ * Exported because the rule, not the code, is the thing worth sharing: `Retry-After` is parsed in
15300
+ * more than one package (fab-pipeline's `getOpenSearchRetryAfterMs`), and a four-token predicate
15301
+ * copied around is a rule that drifts. Feed it a delay already converted to ms.
15302
+ */
15303
+ function retryAfterHintOrNull(ms) {
15304
+ return ms > 0 ? ms : null;
15305
+ }
15306
+ /**
15307
+ * Extract retry delay from error response (e.g., Retry-After header). Returns null when the header is
15308
+ * absent, unparseable, or does not ask us to wait - see retryAfterHintOrNull.
13915
15309
  */
13916
15310
  function getRetryAfterMs(error) {
13917
15311
  if (!isAxiosError(error)) return null;
13918
15312
  const retryAfter = error.response?.headers?.["retry-after"];
13919
15313
  if (!retryAfter) return null;
13920
15314
  const seconds = parseInt(retryAfter, 10);
13921
- if (!isNaN(seconds)) return seconds * 1e3;
15315
+ if (!isNaN(seconds)) return retryAfterHintOrNull(seconds * 1e3);
13922
15316
  const date = Date.parse(retryAfter);
13923
- if (!isNaN(date)) return Math.max(0, date - Date.now());
15317
+ if (!isNaN(date)) return retryAfterHintOrNull(date - Date.now());
13924
15318
  return null;
13925
15319
  }
13926
15320
  /**
@@ -15335,4 +16729,4 @@ var ConfigStore = class {
15335
16729
  }
15336
16730
  };
15337
16731
  //#endregion
15338
- export { VoyageAIEmbeddingModel as $, HTTPError as A, resolveHistoryFetchLimit as At, PermissionDeniedError as B, isNearLimit as Bt, DEFAULT_MUSIC_MODEL_ID as C, isRetryableError as Ct, FIXED_TEMPERATURE_MODELS as D, isZodError as Dt, FIELD_GROUP_OF as E, isUserInitiatedAbort as Et, ModelBackend as F, usdToCredits as Ft, SpeechToTextModels as G, REASONING_SUPPORTED_MODELS as H, NO_TEMPERATURE_MODELS as I, usdToCreditsStochastic as It, TooManyRequestsError as J, SupportedFabFileMimeTypes as K, NotFoundError as L, withRetry as Lt, ImageModels as M, settingsMap as Mt, InternalServerError as N, toModelInfo as Nt, FORMAT_PROMPT_TEMPLATE as O, mapMimeTypeToArtifactType as Ot, MODEL_INFO_FIELD_GROUP_OF as P, toModelRecord as Pt, VideoModels as Q, OllamaEmbeddingModel as R, buildRateLimitLogEntry as Rt, CorruptedFileError as S, isRenderableModelType as St, DEGENERATE_FINISH_REASON as T, isUnlimitedHistory as Tt, REFUSAL_FALLBACK_MODELS as U, REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS as V, parseRateLimitHeaders as Vt, RESPONSES_API_TOOL_MODELS as W, UnprocessableEntityError as X, UnauthorizedError as Y, VIDEO_SIZE_CONSTRAINTS as Z, BadRequestError as _, isImageServeable as _t, getCreditsUrl as a, getMcpProviderMetadata as at, CREDIT_DEDUCT_TRANSACTION_TYPES as b, isModelDeprecated as bt, requireApiUrl as c, isAudioMimeType as ct, AGENT_QUEST_MANIFEST as d, isEarlyStop as dt, WORK_ITEM_STATUSES as et, AGENT_QUEST_MCP_URI as f, isFieldGroup as ft, BFL_SAFETY_TOLERANCE as g, isImageAttachment as gt, BEDROCK_NO_PROMPT_CACHING_MODELS as h, isGeminiModelId as ht, LOCAL_DEV_URL as i, defaultEmbeddingModelForEnv as it, HttpStatus as j, secureParameters as jt, ForbiddenError as k, obfuscateApiKey as kt, resolveApiEndpoint as l, isChunkRebuildPending as lt, ApiKeyType as m, isGPTImageModel as mt, logger as n, calculateRetryDelay as nt, getEnvironmentName as o, getQuestErrorCode as ot, ARTIFACT_ATTRS_PATTERN as p, isGPTImage2Model as pt, TTS_MAX_INPUT_CHARS as q, ApiEndpointUnconfiguredError as r, dayjsConfig_default as rt, parseApiUrl as s, getRetryAfterMs as st, ConfigStore as t, applyModelPriceCatalog as tt, AGENT_QUEST_ID as u, isConvergencePausedNote as ut, BedrockEmbeddingModel as v, isMediaModelType as vt, DEFAULT_UNKNOWN_CONTEXT_WINDOW as w, isSupportedFabFileMimeType as wt, ChatModels as x, isPlaceholderApiKey as xt, CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS as y, isModelAccessible as yt, OpenAIEmbeddingModel as z, extractSnippetMeta as zt };
16732
+ export { SupportedFabFileMimeTypes as $, selfClaimedActorKindSchema as $t, HTTPError as A, isPlaceholderApiKey as At, OPENAI_GPT_IMAGE_1_IMAGE_SIZES as B, reservationOutputTokens as Bt, DEFAULT_MUSIC_MODEL_ID as C, isGPTImageModel as Ct, FIXED_TEMPERATURE_MODELS as D, isMediaModelType as Dt, FIELD_GROUP_OF as E, isImageServeable as Et, MODEL_INFO_FIELD_GROUP_OF as F, isUserInitiatedAbort as Ft, PermissionDeniedError as G, toModelRecord as Gt, OllamaEmbeddingModel as H, secureParameters as Ht, McpServerName as I, isZodError as It, REFUSAL_FALLBACK_MODELS as J, withRetry as Jt, REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS as K, usdToCredits as Kt, ModelBackend as L, mapMimeTypeToArtifactType as Lt, IMAGE_SIZE_CONSTRAINTS as M, isRetryableError as Mt, ImageModels as N, isSupportedFabFileMimeType as Nt, FORMAT_PROMPT_TEMPLATE as O, isModelAccessible as Ot, InternalServerError as P, isUnlimitedHistory as Pt, SpeechToTextModels as Q, actorKindSchema as Qt, NO_TEMPERATURE_MODELS as R, obfuscateApiKey as Rt, CorruptedFileError as S, isGPTImage2Model as St, DEGENERATE_FINISH_REASON as T, isImageAttachment as Tt, OpenAIEmbeddingModel as U, settingsMap as Ut, OPENAI_GPT_IMAGE_2_IMAGE_SIZES as V, resolveHistoryFetchLimit as Vt, PROMPT_TEXT_MAX as W, toModelInfo as Wt, REVIEW_GATE_STATUS_VALUES as X, actorColorIndex as Xt, RESPONSES_API_TOOL_MODELS as Y, ACTOR_COLOR_SLOTS as Yt, SUBQUEST_STATUS_VALUES as Z, actorKindMarker as Zt, BadRequestError as _, isAudioMimeType as _t, getCreditsUrl as a, VideoModels as at, CREDIT_DEDUCT_TRANSACTION_TYPES as b, isEarlyStop as bt, requireApiUrl as c, applyModelPriceCatalog as ct, AGENT_QUEST_MANIFEST as d, defaultEmbeddingModelForEnv as dt, buildRateLimitLogEntry as en, TTS_MAX_INPUT_CHARS as et, AGENT_QUEST_MCP_URI as f, getMcpProviderMetadata as ft, BFL_SAFETY_TOLERANCE as g, hasUsableLimits as gt, BEDROCK_NO_PROMPT_CACHING_MODELS as h, hasKeylessCloudEmbedder as ht, LOCAL_DEV_URL as i, VIDEO_SIZE_CONSTRAINTS as it, HttpStatus as j, isRenderableModelType as jt, ForbiddenError as k, isModelDeprecated as kt, resolveApiEndpoint as l, calculateRetryDelay as lt, ApiKeyType as m, getRetryAfterMs as mt, logger as n, isNearLimit as nn, UnauthorizedError as nt, getEnvironmentName as o, VoyageAIEmbeddingModel as ot, ARTIFACT_ATTRS_PATTERN as p, getQuestErrorCode as pt, REASONING_SUPPORTED_MODELS as q, usdToCreditsStochastic as qt, ApiEndpointUnconfiguredError as r, parseRateLimitHeaders as rn, UnprocessableEntityError as rt, parseApiUrl as s, WORK_ITEM_STATUSES as st, ConfigStore as t, extractSnippetMeta as tn, TooManyRequestsError as tt, AGENT_QUEST_ID as u, dayjsConfig_default as ut, BedrockEmbeddingModel as v, isChunkRebuildPending as vt, DEFAULT_UNKNOWN_CONTEXT_WINDOW as w, isGeminiModelId as wt, ChatModels as x, isFieldGroup as xt, CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS as y, isChunkStalledFile as yt, NotFoundError as z, parseEmbeddingRateLimitHeaders as zt };