@bike4mind/cli 0.20.2 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/{AgentHistoryStore-T7Oh84Yn.mjs → AgentHistoryStore-7zQUOxkA.mjs} +2364 -540
- package/dist/{ApiClient-BopvMmQk.mjs → ApiClient-XdY9O3nz.mjs} +2 -2
- package/dist/{ConfigStore-cIyF7hDg.mjs → ConfigStore-8_0WsN5r.mjs} +1594 -200
- package/dist/{buildAgent-P0tOMLt1.mjs → buildAgent-CvPRH2n-.mjs} +2 -2
- package/dist/commands/acpCommand.mjs +4 -4
- package/dist/commands/apiCommand.mjs +1 -1
- package/dist/commands/doctorCommand.mjs +1 -1
- package/dist/commands/envCommand.mjs +1 -1
- package/dist/commands/headlessCommand.mjs +3 -3
- package/dist/commands/mcpCommand.mjs +3 -3
- package/dist/commands/pluginCommand.mjs +3 -1
- package/dist/commands/updateCommand.mjs +1 -1
- package/dist/index.mjs +13 -54
- package/dist/{package-CnVCHR3U.mjs → package-OZdXqn0e.mjs} +1 -1
- package/dist/{serve-BocVOJ3W.mjs → serve-CEqSZwl6.mjs} +5 -3
- package/package.json +13 -12
|
@@ -7,7 +7,6 @@ import path from "path";
|
|
|
7
7
|
import { v4 } from "uuid";
|
|
8
8
|
import * as z$2 from "zod";
|
|
9
9
|
import z, { ZodError, z as z$1 } from "zod";
|
|
10
|
-
import { actorKindSchema, hearthEventKindSchema, hearthEventRefsSchema, hearthMachineBodySchema } from "@bike4mind/hearth";
|
|
11
10
|
import dayjs from "dayjs";
|
|
12
11
|
import timezone from "dayjs/plugin/timezone.js";
|
|
13
12
|
import utc from "dayjs/plugin/utc.js";
|
|
@@ -52,7 +51,7 @@ const extractSnippetMeta = (content) => {
|
|
|
52
51
|
}
|
|
53
52
|
return { sections };
|
|
54
53
|
};
|
|
55
|
-
function getHeader(headers, name) {
|
|
54
|
+
function getHeader$1(headers, name) {
|
|
56
55
|
if (!headers || typeof headers !== "object") return null;
|
|
57
56
|
if (typeof headers.get === "function") {
|
|
58
57
|
const val = headers.get(name);
|
|
@@ -76,10 +75,10 @@ function parseNumber(value) {
|
|
|
76
75
|
* - `Retry-After` - seconds to wait (on 429 responses), or an HTTP-date
|
|
77
76
|
*/
|
|
78
77
|
function parseRateLimitHeaders(headers) {
|
|
79
|
-
const limitStr = getHeader(headers, "X-RateLimit-Limit") ?? getHeader(headers, "x-ratelimit-limit");
|
|
80
|
-
const remainingStr = getHeader(headers, "X-RateLimit-Remaining") ?? getHeader(headers, "x-ratelimit-remaining");
|
|
81
|
-
const resetStr = getHeader(headers, "X-RateLimit-Reset") ?? getHeader(headers, "x-ratelimit-reset");
|
|
82
|
-
const retryAfterStr = getHeader(headers, "Retry-After") ?? getHeader(headers, "retry-after");
|
|
78
|
+
const limitStr = getHeader$1(headers, "X-RateLimit-Limit") ?? getHeader$1(headers, "x-ratelimit-limit");
|
|
79
|
+
const remainingStr = getHeader$1(headers, "X-RateLimit-Remaining") ?? getHeader$1(headers, "x-ratelimit-remaining");
|
|
80
|
+
const resetStr = getHeader$1(headers, "X-RateLimit-Reset") ?? getHeader$1(headers, "x-ratelimit-reset");
|
|
81
|
+
const retryAfterStr = getHeader$1(headers, "Retry-After") ?? getHeader$1(headers, "retry-after");
|
|
83
82
|
const limit = parseNumber(limitStr);
|
|
84
83
|
const remaining = parseNumber(remainingStr);
|
|
85
84
|
let resetAt = null;
|
|
@@ -110,13 +109,6 @@ function parseRateLimitHeaders(headers) {
|
|
|
110
109
|
usagePercent
|
|
111
110
|
};
|
|
112
111
|
}
|
|
113
|
-
/**
|
|
114
|
-
* Check whether the current rate limit usage is near the threshold.
|
|
115
|
-
*
|
|
116
|
-
* @param info - Parsed rate limit info
|
|
117
|
-
* @param thresholdPercent - Usage percentage threshold (default: 80)
|
|
118
|
-
* @returns true if usage is at or above the threshold
|
|
119
|
-
*/
|
|
120
112
|
function isNearLimit(info, thresholdPercent = 80) {
|
|
121
113
|
if (info.usagePercent === null) return false;
|
|
122
114
|
return info.usagePercent >= thresholdPercent;
|
|
@@ -140,6 +132,151 @@ function buildRateLimitLogEntry(integration, endpoint, info, wasThrottled = fals
|
|
|
140
132
|
};
|
|
141
133
|
}
|
|
142
134
|
//#endregion
|
|
135
|
+
//#region ../../b4m-core/hearth/dist/index.mjs
|
|
136
|
+
/**
|
|
137
|
+
* Zod validation for data crossing the Hearth boundary (API routes, CLI
|
|
138
|
+
* tools, gateways). Must stay in sync with the types in types.ts.
|
|
139
|
+
*/
|
|
140
|
+
const actorKindSchema = z$1.enum([
|
|
141
|
+
"human",
|
|
142
|
+
"agent",
|
|
143
|
+
"gateway",
|
|
144
|
+
"device",
|
|
145
|
+
"system"
|
|
146
|
+
]);
|
|
147
|
+
/**
|
|
148
|
+
* The kinds a caller may claim FOR ITSELF. 'human' and 'system' are reserved:
|
|
149
|
+
* the human actor is derived from the authenticated session, never from a
|
|
150
|
+
* request body, so no credential can post an event that renders as the account
|
|
151
|
+
* owner. Claiming one of these three is a downgrade in trust, not a spoof.
|
|
152
|
+
*
|
|
153
|
+
* Single source for both self-identification paths - the actor override and the
|
|
154
|
+
* per-session kind (see HearthActorParamSchema / HearthSessionParamSchema in
|
|
155
|
+
* the client's hearthWire) - so the reserved set cannot drift between them.
|
|
156
|
+
*/
|
|
157
|
+
const selfClaimedActorKindSchema = z$1.enum([
|
|
158
|
+
"agent",
|
|
159
|
+
"gateway",
|
|
160
|
+
"device"
|
|
161
|
+
]);
|
|
162
|
+
const hearthEventKindSchema = z$1.enum([
|
|
163
|
+
"message",
|
|
164
|
+
"edit",
|
|
165
|
+
"reaction",
|
|
166
|
+
"artifact",
|
|
167
|
+
"presence",
|
|
168
|
+
"delegation",
|
|
169
|
+
"quest.update",
|
|
170
|
+
"gate.request",
|
|
171
|
+
"gate.resolve",
|
|
172
|
+
"system"
|
|
173
|
+
]);
|
|
174
|
+
const hearthHumanBodySchema = z$1.object({
|
|
175
|
+
text: z$1.string().min(1),
|
|
176
|
+
format: z$1.enum(["md", "text"])
|
|
177
|
+
});
|
|
178
|
+
const hearthMachineBodySchema = z$1.object({
|
|
179
|
+
schema: z$1.string().min(1),
|
|
180
|
+
payload: z$1.unknown()
|
|
181
|
+
});
|
|
182
|
+
const hearthEventRefsSchema = z$1.object({
|
|
183
|
+
threadRootId: z$1.string().min(1).optional(),
|
|
184
|
+
replyToId: z$1.string().min(1).optional(),
|
|
185
|
+
questId: z$1.string().min(1).optional(),
|
|
186
|
+
externalId: z$1.string().min(1).optional()
|
|
187
|
+
});
|
|
188
|
+
z$1.object({
|
|
189
|
+
channelId: z$1.string().min(1),
|
|
190
|
+
actorId: z$1.string().min(1),
|
|
191
|
+
kind: hearthEventKindSchema,
|
|
192
|
+
human: hearthHumanBodySchema,
|
|
193
|
+
machine: hearthMachineBodySchema.optional(),
|
|
194
|
+
refs: hearthEventRefsSchema
|
|
195
|
+
});
|
|
196
|
+
/** djb2. Deterministic across processes and restarts, which is the whole point. */
|
|
197
|
+
function hashOf(value) {
|
|
198
|
+
let hash = 5381;
|
|
199
|
+
for (let i = 0; i < value.length; i++) hash = (hash << 5) + hash + value.charCodeAt(i) | 0;
|
|
200
|
+
return Math.abs(hash);
|
|
201
|
+
}
|
|
202
|
+
/**
|
|
203
|
+
* Stable palette slot for an actor. Hash-derived, never array index or arrival
|
|
204
|
+
* order: those repaint every actor whenever the tail changes or a reload
|
|
205
|
+
* reorders the buffer, and a shifting color is worse than no color for telling
|
|
206
|
+
* two agents apart.
|
|
207
|
+
*/
|
|
208
|
+
function actorColorIndex(actorId) {
|
|
209
|
+
if (!actorId) return 0;
|
|
210
|
+
return hashOf(actorId) % 4;
|
|
211
|
+
}
|
|
212
|
+
const ACTOR_COLOR_SLOTS = [
|
|
213
|
+
{
|
|
214
|
+
light: "#2a78d6",
|
|
215
|
+
dark: "#3987e5"
|
|
216
|
+
},
|
|
217
|
+
{
|
|
218
|
+
light: "#eda100",
|
|
219
|
+
dark: "#c98500"
|
|
220
|
+
},
|
|
221
|
+
{
|
|
222
|
+
light: "#e87ba4",
|
|
223
|
+
dark: "#d55181"
|
|
224
|
+
},
|
|
225
|
+
{
|
|
226
|
+
light: "#008300",
|
|
227
|
+
dark: "#008300"
|
|
228
|
+
}
|
|
229
|
+
];
|
|
230
|
+
/** Single-character marker for text-only surfaces (the CLI /hearth listing). */
|
|
231
|
+
const ACTOR_KIND_MARKERS = {
|
|
232
|
+
human: "H",
|
|
233
|
+
agent: "A",
|
|
234
|
+
gateway: "G",
|
|
235
|
+
device: "D",
|
|
236
|
+
system: "S"
|
|
237
|
+
};
|
|
238
|
+
/**
|
|
239
|
+
* Read through a Partial view on purpose. The type is closed within one build,
|
|
240
|
+
* but the SPA bare-casts the WS payload, so a client running stale code against
|
|
241
|
+
* a server that has added a kind receives one that is absent from these maps -
|
|
242
|
+
* and a blank badge reads as "no kind stated" rather than "kind unknown", which
|
|
243
|
+
* is the opposite of what the mitigation needs to say.
|
|
244
|
+
*/
|
|
245
|
+
function lookup(map, kind, fallback) {
|
|
246
|
+
if (!kind) return fallback;
|
|
247
|
+
return map[kind] ?? fallback;
|
|
248
|
+
}
|
|
249
|
+
function actorKindMarker(kind) {
|
|
250
|
+
return lookup(ACTOR_KIND_MARKERS, kind, "?");
|
|
251
|
+
}
|
|
252
|
+
z$1.object({
|
|
253
|
+
/** Claude Code lifecycle event name; the reason fallback for tiers 0 and 1. */
|
|
254
|
+
hook_event_name: z$1.string().nullish(),
|
|
255
|
+
session_id: z$1.string().nullish(),
|
|
256
|
+
slug: z$1.string().nullish(),
|
|
257
|
+
/** Workspace BASENAME. Never a full path - that is content. */
|
|
258
|
+
workspace: z$1.string().nullish(),
|
|
259
|
+
/**
|
|
260
|
+
* Which reporter wrote this. A LOOSE string, not an enum: a fourth surface
|
|
261
|
+
* posting an unrecognized value must land on the roster with its detail
|
|
262
|
+
* intact, and a strict enum would fail the whole parse and drop the row - the
|
|
263
|
+
* exact "a third reporter is expensive" cost this contract exists to remove.
|
|
264
|
+
*/
|
|
265
|
+
surface: z$1.string().nullish(),
|
|
266
|
+
/** Engine driving the session, where the reporter knows it. */
|
|
267
|
+
source: z$1.string().nullish(),
|
|
268
|
+
claude_version: z$1.string().nullish(),
|
|
269
|
+
activity: z$1.object({
|
|
270
|
+
reason: z$1.string().nullish(),
|
|
271
|
+
tool: z$1.string().nullish(),
|
|
272
|
+
permission_mode: z$1.string().nullish(),
|
|
273
|
+
effort: z$1.string().nullish(),
|
|
274
|
+
duration_ms: z$1.number().nullish(),
|
|
275
|
+
subagent: z$1.string().nullish(),
|
|
276
|
+
background_tasks: z$1.number().int().min(0).nullish()
|
|
277
|
+
}).nullish()
|
|
278
|
+
});
|
|
279
|
+
//#endregion
|
|
143
280
|
//#region ../../b4m-core/common/dist/index.mjs
|
|
144
281
|
let HttpStatus = /* @__PURE__ */ function(HttpStatus) {
|
|
145
282
|
HttpStatus[HttpStatus["Ok"] = 200] = "Ok";
|
|
@@ -148,9 +285,11 @@ let HttpStatus = /* @__PURE__ */ function(HttpStatus) {
|
|
|
148
285
|
HttpStatus[HttpStatus["Unauthorized"] = 401] = "Unauthorized";
|
|
149
286
|
HttpStatus[HttpStatus["Forbidden"] = 403] = "Forbidden";
|
|
150
287
|
HttpStatus[HttpStatus["NotFound"] = 404] = "NotFound";
|
|
288
|
+
HttpStatus[HttpStatus["Conflict"] = 409] = "Conflict";
|
|
151
289
|
HttpStatus[HttpStatus["UnprocessableEntity"] = 422] = "UnprocessableEntity";
|
|
152
290
|
HttpStatus[HttpStatus["TooManyRequests"] = 429] = "TooManyRequests";
|
|
153
291
|
HttpStatus[HttpStatus["InternalServerError"] = 500] = "InternalServerError";
|
|
292
|
+
HttpStatus[HttpStatus["BadGateway"] = 502] = "BadGateway";
|
|
154
293
|
return HttpStatus;
|
|
155
294
|
}({});
|
|
156
295
|
var HTTPError = class extends Error {
|
|
@@ -277,6 +416,7 @@ let ApiKeyType = /* @__PURE__ */ function(ApiKeyType) {
|
|
|
277
416
|
ApiKeyType["ollama"] = "ollama";
|
|
278
417
|
ApiKeyType["xai"] = "xai";
|
|
279
418
|
ApiKeyType["kimi"] = "kimi";
|
|
419
|
+
ApiKeyType["deepseek"] = "deepseek";
|
|
280
420
|
ApiKeyType["voyageai"] = "voyageai";
|
|
281
421
|
return ApiKeyType;
|
|
282
422
|
}({});
|
|
@@ -312,6 +452,7 @@ const b4mLLMTools = z$1.enum([
|
|
|
312
452
|
"chess_engine",
|
|
313
453
|
"retrieve_knowledge_content",
|
|
314
454
|
"count_knowledge_base",
|
|
455
|
+
"describe_knowledge_base",
|
|
315
456
|
"delegate_to_agent",
|
|
316
457
|
"optihashi_schedule",
|
|
317
458
|
"optihashi_formulate",
|
|
@@ -349,6 +490,16 @@ const FallbackInfoSchema = z$1.object({
|
|
|
349
490
|
primaryModelName: z$1.string(),
|
|
350
491
|
fallbackModel: z$1.string(),
|
|
351
492
|
fallbackModelName: z$1.string(),
|
|
493
|
+
/**
|
|
494
|
+
* Backend of each side, so the badge can name the provider path without
|
|
495
|
+
* inferring it from the id - a bare `deepseek-*` slug is the direct API on a
|
|
496
|
+
* hosted deployment and an Ollama pull on a self-hosted one. Optional and
|
|
497
|
+
* loosely typed: these payloads are persisted in browser storage, so a record
|
|
498
|
+
* written by an older build has neither field and an unknown backend name from
|
|
499
|
+
* a newer one must not fail the parse.
|
|
500
|
+
*/
|
|
501
|
+
primaryModelBackend: z$1.string().optional(),
|
|
502
|
+
fallbackModelBackend: z$1.string().optional(),
|
|
352
503
|
timestamp: z$1.number()
|
|
353
504
|
});
|
|
354
505
|
const ArtifactMetadataSchema = z$1.object({
|
|
@@ -755,6 +906,7 @@ let ModelBackend = /* @__PURE__ */ function(ModelBackend) {
|
|
|
755
906
|
ModelBackend["BFL"] = "bfl";
|
|
756
907
|
ModelBackend["XAI"] = "xai";
|
|
757
908
|
ModelBackend["Kimi"] = "kimi";
|
|
909
|
+
ModelBackend["DeepSeek"] = "deepseek";
|
|
758
910
|
ModelBackend["VoyageAI"] = "voyageai";
|
|
759
911
|
ModelBackend["AWS"] = "aws";
|
|
760
912
|
ModelBackend["LocalImage"] = "local-image";
|
|
@@ -785,6 +937,62 @@ let ImageModels = /* @__PURE__ */ function(ImageModels) {
|
|
|
785
937
|
Object.values(ImageModels);
|
|
786
938
|
z$1.enum(ImageModels);
|
|
787
939
|
/**
|
|
940
|
+
* Image size constraints and options
|
|
941
|
+
*/
|
|
942
|
+
const IMAGE_SIZE_CONSTRAINTS = {
|
|
943
|
+
BFL: {
|
|
944
|
+
minWidth: 256,
|
|
945
|
+
maxWidth: 1440,
|
|
946
|
+
minHeight: 256,
|
|
947
|
+
maxHeight: 1440,
|
|
948
|
+
stepSize: 32,
|
|
949
|
+
defaultSize: "1280x960",
|
|
950
|
+
sizes: [
|
|
951
|
+
"1280x960",
|
|
952
|
+
"1024x768",
|
|
953
|
+
"800x600",
|
|
954
|
+
"1280x720",
|
|
955
|
+
"1024x576",
|
|
956
|
+
"1440x810",
|
|
957
|
+
"1024x1024",
|
|
958
|
+
"768x768",
|
|
959
|
+
"512x512",
|
|
960
|
+
"960x1280",
|
|
961
|
+
"768x1024",
|
|
962
|
+
"600x800"
|
|
963
|
+
]
|
|
964
|
+
},
|
|
965
|
+
GPT_IMAGE_1: {
|
|
966
|
+
sizes: [
|
|
967
|
+
"1024x1024",
|
|
968
|
+
"1024x1536",
|
|
969
|
+
"1536x1024"
|
|
970
|
+
],
|
|
971
|
+
defaultSize: "1024x1024"
|
|
972
|
+
},
|
|
973
|
+
GPT_IMAGE_2: {
|
|
974
|
+
/** Popular preset sizes shown in the UI. The API accepts any resolution meeting the constraints. */
|
|
975
|
+
sizes: [
|
|
976
|
+
"1024x1024",
|
|
977
|
+
"1536x1024",
|
|
978
|
+
"1024x1536",
|
|
979
|
+
"2048x2048",
|
|
980
|
+
"2048x1152",
|
|
981
|
+
"3840x2160",
|
|
982
|
+
"2160x3840"
|
|
983
|
+
],
|
|
984
|
+
defaultSize: "1024x1024",
|
|
985
|
+
/** Constraints for custom/flexible sizes */
|
|
986
|
+
constraints: {
|
|
987
|
+
maxEdge: 3840,
|
|
988
|
+
minTotalPixels: 655360,
|
|
989
|
+
maxTotalPixels: 8294400,
|
|
990
|
+
edgeMultiple: 16,
|
|
991
|
+
maxAspectRatio: 3
|
|
992
|
+
}
|
|
993
|
+
}
|
|
994
|
+
};
|
|
995
|
+
/**
|
|
788
996
|
* Chat Models
|
|
789
997
|
*
|
|
790
998
|
* https://platform.openai.com/docs/models/continuous-model-upgrades
|
|
@@ -897,6 +1105,7 @@ let ChatModels = /* @__PURE__ */ function(ChatModels) {
|
|
|
897
1105
|
ChatModels["KIMI_K2_5"] = "kimi-k2.5";
|
|
898
1106
|
ChatModels["KIMI_K2_5_BEDROCK"] = "moonshotai.kimi-k2.5";
|
|
899
1107
|
ChatModels["KIMI_K2_THINKING_BEDROCK"] = "moonshot.kimi-k2-thinking";
|
|
1108
|
+
ChatModels["DEEPSEEK_FLASH"] = "deepseek-flash";
|
|
900
1109
|
return ChatModels;
|
|
901
1110
|
}({});
|
|
902
1111
|
const CHAT_MODELS = Object.values(ChatModels);
|
|
@@ -1031,7 +1240,8 @@ const NO_TEMPERATURE_MODELS = /* @__PURE__ */ new Set([
|
|
|
1031
1240
|
"kimi-k2.7-code",
|
|
1032
1241
|
"kimi-k2.7-code-highspeed",
|
|
1033
1242
|
"kimi-k2.6",
|
|
1034
|
-
"kimi-k2.5"
|
|
1243
|
+
"kimi-k2.5",
|
|
1244
|
+
"deepseek-flash"
|
|
1035
1245
|
]);
|
|
1036
1246
|
/**
|
|
1037
1247
|
* Models whose safety classifiers can decline a request with `stop_reason: 'refusal'`
|
|
@@ -1122,6 +1332,14 @@ const SUBQUEST_STATUS_VALUES = [
|
|
|
1122
1332
|
"deleted"
|
|
1123
1333
|
];
|
|
1124
1334
|
/**
|
|
1335
|
+
* Status of a review gate on a sub-quest
|
|
1336
|
+
*/
|
|
1337
|
+
const REVIEW_GATE_STATUS_VALUES = [
|
|
1338
|
+
"pending",
|
|
1339
|
+
"approved",
|
|
1340
|
+
"rejected"
|
|
1341
|
+
];
|
|
1342
|
+
/**
|
|
1125
1343
|
* Valid complexity ratings for quests.
|
|
1126
1344
|
* Canonical vocabulary - matches what the planner generates and validates
|
|
1127
1345
|
* (Easy < 1 hour, Medium 1-4 hours, Hard > 4 hours).
|
|
@@ -1512,6 +1730,7 @@ const ADAPTER_FAMILIES = [
|
|
|
1512
1730
|
"gemini",
|
|
1513
1731
|
"xai",
|
|
1514
1732
|
"kimi",
|
|
1733
|
+
"deepseek",
|
|
1515
1734
|
"ollama",
|
|
1516
1735
|
"bfl",
|
|
1517
1736
|
"local-image",
|
|
@@ -1762,6 +1981,7 @@ const MODEL_INFO_FIELD_GROUP_OF = {
|
|
|
1762
1981
|
supportsImageVariation: "modalities",
|
|
1763
1982
|
supportsSafetyTolerance: "modalities",
|
|
1764
1983
|
deprecationDate: "lifecycle",
|
|
1984
|
+
replacedBy: "lifecycle",
|
|
1765
1985
|
adapterFamily: "dispatch",
|
|
1766
1986
|
dispatchProfile: "dispatch",
|
|
1767
1987
|
description: "presentation",
|
|
@@ -1885,6 +2105,8 @@ z$1.object({
|
|
|
1885
2105
|
}).optional(),
|
|
1886
2106
|
/** Optional: a state row written before Phase 4 has none, and must keep parsing. */
|
|
1887
2107
|
suggestion: ModelLifecycleSuggestion.optional(),
|
|
2108
|
+
/** Dispatch probes spent on this model, so one that always fails stops taking a budget slot. */
|
|
2109
|
+
probeAttempts: z$1.number().int().nonnegative().optional(),
|
|
1888
2110
|
createdAt: z$1.date(),
|
|
1889
2111
|
updatedAt: z$1.date()
|
|
1890
2112
|
});
|
|
@@ -2188,6 +2410,13 @@ z$1.object({
|
|
|
2188
2410
|
createdAt: z$1.date(),
|
|
2189
2411
|
updatedAt: z$1.date()
|
|
2190
2412
|
});
|
|
2413
|
+
let McpServerName = /* @__PURE__ */ function(McpServerName) {
|
|
2414
|
+
McpServerName["LinkedIn"] = "linkedin";
|
|
2415
|
+
McpServerName["Github"] = "github";
|
|
2416
|
+
McpServerName["Atlassian"] = "atlassian";
|
|
2417
|
+
McpServerName["Notion"] = "notion";
|
|
2418
|
+
return McpServerName;
|
|
2419
|
+
}({});
|
|
2191
2420
|
/**
|
|
2192
2421
|
* Check if a value is a placeholder (not configured or SST default).
|
|
2193
2422
|
* Uses case-insensitive comparison and trims whitespace to prevent bypass attempts.
|
|
@@ -2233,6 +2462,42 @@ function isPlaceholderApiKey(value) {
|
|
|
2233
2462
|
if (!normalized) return true;
|
|
2234
2463
|
return PLACEHOLDER_API_KEY_REGEX.test(normalized);
|
|
2235
2464
|
}
|
|
2465
|
+
/**
|
|
2466
|
+
* Lake lifecycle. Stable states (draft/active/archived/deleted) plus transitional
|
|
2467
|
+
* states (archiving/unarchiving/restoring/deleting/purging) that exist to drive UI and make a crashed
|
|
2468
|
+
* mid-operation observable. draft -> active is one-way. It happens implicitly once the lake
|
|
2469
|
+
* holds its first member file (see `activateIfDraft` below), and unconditionally when an
|
|
2470
|
+
* archived or deleted lake is restored, which is how an empty lake can end up active.
|
|
2471
|
+
*
|
|
2472
|
+
* `purging` is the one transitional state that is NOT recoverable by retrying the same action:
|
|
2473
|
+
* it is claimed the moment a phase-2 hard delete is ACCEPTED (#1744), before the background
|
|
2474
|
+
* sweep runs, so that `listDeletedDataLakes` stops offering Restore on a lake whose
|
|
2475
|
+
* destruction is already irreversible. Everything else that reads `status` must treat it as
|
|
2476
|
+
* "going away", never as a lake to act on.
|
|
2477
|
+
*/
|
|
2478
|
+
const DATA_LAKE_STATUSES = [
|
|
2479
|
+
"draft",
|
|
2480
|
+
"active",
|
|
2481
|
+
"archiving",
|
|
2482
|
+
"archived",
|
|
2483
|
+
"unarchiving",
|
|
2484
|
+
"restoring",
|
|
2485
|
+
"deleting",
|
|
2486
|
+
"deleted",
|
|
2487
|
+
"purging"
|
|
2488
|
+
];
|
|
2489
|
+
/**
|
|
2490
|
+
* Stable (non-transitional) lake statuses - a lake sitting in one of these is at rest, not
|
|
2491
|
+
* mid-operation. Load-bearing as the INPUT to `DATA_LAKE_TRANSITIONAL_STATUSES` below, which is
|
|
2492
|
+
* what drives the needs-attention list; it is not itself a filter any list path applies.
|
|
2493
|
+
*/
|
|
2494
|
+
const DATA_LAKE_STABLE_STATUSES = [
|
|
2495
|
+
"draft",
|
|
2496
|
+
"active",
|
|
2497
|
+
"archived",
|
|
2498
|
+
"deleted"
|
|
2499
|
+
];
|
|
2500
|
+
DATA_LAKE_STATUSES.filter((s) => !DATA_LAKE_STABLE_STATUSES.includes(s));
|
|
2236
2501
|
z$1.object({
|
|
2237
2502
|
/**
|
|
2238
2503
|
* The granting lake's Mongo `_id`. ALWAYS a persisted DB lake: a hardcoded/fallback lake has no
|
|
@@ -2263,6 +2528,23 @@ z$1.object({
|
|
|
2263
2528
|
*/
|
|
2264
2529
|
expiresAt: z$1.date().nullish()
|
|
2265
2530
|
});
|
|
2531
|
+
/** One prefix-arm content tag a lake held on a file, as captured at removal for later restore. */
|
|
2532
|
+
const LakeMembershipRemovalContentTag = z$1.object({
|
|
2533
|
+
name: z$1.string(),
|
|
2534
|
+
strength: z$1.number()
|
|
2535
|
+
});
|
|
2536
|
+
z$1.object({
|
|
2537
|
+
dataLakeId: z$1.string(),
|
|
2538
|
+
fabFileId: z$1.string(),
|
|
2539
|
+
/** The principal who performed the removal - audit only, never a restore authorization input. */
|
|
2540
|
+
actorUserId: z$1.string(),
|
|
2541
|
+
/** The lake's prefix-arm tags the file carried, captured by `lakeMembershipSignals`. May be
|
|
2542
|
+
* empty - a meta-tag-only member and a non-creator-owned file both removed with no content
|
|
2543
|
+
* tags to restore, which is a complete and legitimate answer, not "nothing to record". */
|
|
2544
|
+
contentTags: z$1.array(LakeMembershipRemovalContentTag),
|
|
2545
|
+
removedAt: z$1.date(),
|
|
2546
|
+
expiresAt: z$1.date()
|
|
2547
|
+
});
|
|
2266
2548
|
/**
|
|
2267
2549
|
* Every `IDataLake` field, classified as audited or not. A TOTAL map keyed by `keyof IDataLake`,
|
|
2268
2550
|
* exactly like `LAKE_FIELD_VISIBILITY` in redactLakeForActor.ts and for the same reason: a list of
|
|
@@ -2289,6 +2571,7 @@ const LAKE_CONFIG_FIELD_AUDIT = {
|
|
|
2289
2571
|
organizationId: "audited",
|
|
2290
2572
|
isPublic: "audited",
|
|
2291
2573
|
auditQueryTextEnabled: "audited",
|
|
2574
|
+
lakeMemoryEnabled: "audited",
|
|
2292
2575
|
status: "audited",
|
|
2293
2576
|
createdByUserId: "audited",
|
|
2294
2577
|
lastUpdatedByUserId: "excluded",
|
|
@@ -2300,7 +2583,10 @@ const LAKE_CONFIG_FIELD_AUDIT = {
|
|
|
2300
2583
|
filesDeletedAt: "excluded",
|
|
2301
2584
|
filesArchivedAt: "excluded",
|
|
2302
2585
|
lakeMemoryExtractionAt: "excluded",
|
|
2303
|
-
lakeMemoryCursor: "excluded"
|
|
2586
|
+
lakeMemoryCursor: "excluded",
|
|
2587
|
+
lakeMemoryPurgedAt: "excluded",
|
|
2588
|
+
inconsistencyReport: "excluded",
|
|
2589
|
+
inconsistencyComputedAt: "excluded"
|
|
2304
2590
|
};
|
|
2305
2591
|
[...Object.keys(LAKE_CONFIG_FIELD_AUDIT).filter((field) => LAKE_CONFIG_FIELD_AUDIT[field] === "audited")];
|
|
2306
2592
|
/**
|
|
@@ -2862,6 +3148,7 @@ function toModelInfo(record) {
|
|
|
2862
3148
|
disabled,
|
|
2863
3149
|
disabledReason: record.disabledReason ?? record.autoDisabledReason ?? (retired ? "retired by the provider" : void 0),
|
|
2864
3150
|
deprecationDate: record.lifecycle?.deprecationDate,
|
|
3151
|
+
replacedBy: record.lifecycle?.replacedBy,
|
|
2865
3152
|
trainingCutoff: record.trainingCutoff,
|
|
2866
3153
|
releaseDate: record.releaseDate,
|
|
2867
3154
|
logoFile: record.logoFile,
|
|
@@ -2894,6 +3181,7 @@ const VENDOR_BY_BACKEND = {
|
|
|
2894
3181
|
["gemini"]: "google",
|
|
2895
3182
|
["xai"]: "xai",
|
|
2896
3183
|
["kimi"]: "moonshotai",
|
|
3184
|
+
["deepseek"]: "deepseek",
|
|
2897
3185
|
["bfl"]: "black-forest-labs",
|
|
2898
3186
|
["aws"]: "amazon",
|
|
2899
3187
|
["voyageai"]: "voyageai",
|
|
@@ -2962,10 +3250,13 @@ function toModelRecord(info) {
|
|
|
2962
3250
|
supportsTools: info.supportsTools,
|
|
2963
3251
|
supportsImageVariation: info.supportsImageVariation,
|
|
2964
3252
|
supportsSafetyTolerance: info.supportsSafetyTolerance,
|
|
2965
|
-
lifecycle:
|
|
2966
|
-
|
|
2967
|
-
|
|
2968
|
-
|
|
3253
|
+
lifecycle: {
|
|
3254
|
+
...info.deprecationDate ? {
|
|
3255
|
+
status: "deprecated",
|
|
3256
|
+
deprecationDate: info.deprecationDate
|
|
3257
|
+
} : { status: "active" },
|
|
3258
|
+
...info.replacedBy ? { replacedBy: info.replacedBy } : {}
|
|
3259
|
+
},
|
|
2969
3260
|
description: info.description,
|
|
2970
3261
|
logoFile: info.logoFile,
|
|
2971
3262
|
rank: info.rank,
|
|
@@ -3094,6 +3385,71 @@ const usdToCreditsStochastic = (usd, rng = cryptoUniform, rate = CREDITS_PER_USD
|
|
|
3094
3385
|
const fraction = raw - base;
|
|
3095
3386
|
return base + (rng() < fraction ? 1 : 0);
|
|
3096
3387
|
};
|
|
3388
|
+
/**
|
|
3389
|
+
* Output tokens the pre-flight credit hold prices when a request's max_tokens
|
|
3390
|
+
* ceiling exceeds it.
|
|
3391
|
+
*
|
|
3392
|
+
* max_tokens is a *ceiling*, not a prediction: adaptive reasoning models stop at
|
|
3393
|
+
* end_turn well short of it, so pricing the hold at the full window (128K on the
|
|
3394
|
+
* flagship models) reserved several thousand credits per turn regardless of answer
|
|
3395
|
+
* length. The excess was always refunded at settlement, but the hold IS the
|
|
3396
|
+
* insufficient-funds gate, so users whose balance sat between their real cost and
|
|
3397
|
+
* the worst case were falsely blocked.
|
|
3398
|
+
*
|
|
3399
|
+
* 16K covers the realistic long answer with headroom - the largest replies seen in
|
|
3400
|
+
* practice are HTML-artifact turns at roughly 10-11K output tokens (see
|
|
3401
|
+
* buildThinkingParams in llm-adapters/thinkingParams.ts, whose max_tokens floor was
|
|
3402
|
+
* sized off the same measurement).
|
|
3403
|
+
*
|
|
3404
|
+
* UNDER-RESERVATION IS ACCEPTED, NOT PREVENTED. A turn that emits more than this
|
|
3405
|
+
* settles as a shortfall debit at reconciliation, which is already a supported path
|
|
3406
|
+
* (provider-basis settlement could always exceed a hold priced on the local
|
|
3407
|
+
* estimate). The two sites clamp the shortfall differently: chat's
|
|
3408
|
+
* computeSettlementDelta (services/llm/ChatCompletionProcess.ts) floors the debit at
|
|
3409
|
+
* the balance snapshot taken at its OWN admission and reports the remainder as
|
|
3410
|
+
* writtenOffCredits; cliCompletions.ts does an unclamped $inc on success and only
|
|
3411
|
+
* logs an ALERT if it lands negative. Neither is a non-negativity guarantee: the
|
|
3412
|
+
* chat snapshot predates any sibling turn's spend (see that function's own "best
|
|
3413
|
+
* effort" note), and when the shortfall still fits the stale snapshot the debit
|
|
3414
|
+
* applies in full - writtenOffCredits 0, and no BILLING_SHORTFALL_CLAMP log - so a
|
|
3415
|
+
* concurrent holder can land negative there too. Either way the resulting balance
|
|
3416
|
+
* fails the *next* turn's own admission gate - but that bound is per turn, not per
|
|
3417
|
+
* holder: turns admitted concurrently are each checked against the balance at their
|
|
3418
|
+
* own admission and settle against a snapshot that predates their siblings' spend,
|
|
3419
|
+
* so a holder running turns in parallel can be shorted once per in-flight turn, not
|
|
3420
|
+
* once total.
|
|
3421
|
+
*/
|
|
3422
|
+
const PREFLIGHT_RESERVATION_OUTPUT_TOKENS = 16384;
|
|
3423
|
+
/**
|
|
3424
|
+
* Reservation ceiling for models that spend reasoning tokens inside their output
|
|
3425
|
+
* budget (see reasonsWithinOutputBudget in llm-adapters/thinkingParams.ts). Those
|
|
3426
|
+
* tokens bill as output on top of the visible answer, so the 16K figure above -
|
|
3427
|
+
* which was measured on visible artifact size alone - under-reserves them badly.
|
|
3428
|
+
*
|
|
3429
|
+
* Deliberately below ADAPTIVE_THINKING_MAX_TOKENS_FLOOR (64K), which sizes the
|
|
3430
|
+
* request's real ceiling and therefore has to cover the worst case: exceeding it
|
|
3431
|
+
* truncates a reply mid-tag, an unrecoverable failure. A hold has no such duty -
|
|
3432
|
+
* exceeding it settles as a shortfall debit - so it is sized for the long turn
|
|
3433
|
+
* (roughly 3x the largest observed visible answer, leaving the rest for the trace)
|
|
3434
|
+
* rather than the worst one, which is what keeps the gate off affordable requests.
|
|
3435
|
+
*/
|
|
3436
|
+
const PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS = 32768;
|
|
3437
|
+
/**
|
|
3438
|
+
* Output-token figure to price a pre-flight credit hold at, given the max_tokens
|
|
3439
|
+
* this request will actually send. Never raises the caller's ceiling: a request
|
|
3440
|
+
* that asks for less than the cap holds only what it can possibly spend.
|
|
3441
|
+
*
|
|
3442
|
+
* Reserving only, never gating: the per-member org credit cap is still priced on
|
|
3443
|
+
* the unshrunk ceiling at both call sites, since that check has no settlement
|
|
3444
|
+
* counterpart to correct an under-estimate. That is a strictly larger figure than
|
|
3445
|
+
* the hold, not a true upper bound on the turn - it prices one model round trip,
|
|
3446
|
+
* at the uncached input rate, on the primary model - so it still under-counts a
|
|
3447
|
+
* multi-round tool loop, a cache-write turn, or a fallback hop onto pricier pricing.
|
|
3448
|
+
*
|
|
3449
|
+
* @param reasonsWithinOutputBudget - reasonsWithinOutputBudget(modelInfo); passed as
|
|
3450
|
+
* a boolean because common cannot import llm-adapters.
|
|
3451
|
+
*/
|
|
3452
|
+
const reservationOutputTokens = (requestedMaxTokens, reasonsWithinOutputBudget = false) => Math.min(requestedMaxTokens, reasonsWithinOutputBudget ? PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS : PREFLIGHT_RESERVATION_OUTPUT_TOKENS);
|
|
3097
3453
|
z.enum([
|
|
3098
3454
|
"openai",
|
|
3099
3455
|
"test",
|
|
@@ -3101,6 +3457,58 @@ z.enum([
|
|
|
3101
3457
|
"xai",
|
|
3102
3458
|
"gemini"
|
|
3103
3459
|
]);
|
|
3460
|
+
z$1.enum([
|
|
3461
|
+
"inject",
|
|
3462
|
+
"auto-fire",
|
|
3463
|
+
"hidden"
|
|
3464
|
+
]);
|
|
3465
|
+
/**
|
|
3466
|
+
* Modes acceptable at AUTHORING time. 'hidden' is intentionally excluded until
|
|
3467
|
+
* the host has true hidden-send support - accepting it would persist a value
|
|
3468
|
+
* that silently behaves as 'auto-fire' (a surprising downgrade). It stays in
|
|
3469
|
+
* ExecutionModeSchema/the stored enum for forward-compat.
|
|
3470
|
+
*/
|
|
3471
|
+
const AuthorableExecutionModeSchema = z$1.enum(["inject", "auto-fire"]);
|
|
3472
|
+
/**
|
|
3473
|
+
* Tools a prompt may require - constrained to the host's closed tool set, MINUS
|
|
3474
|
+
* integration-gated tools that act on the caller's own credentials/account. A
|
|
3475
|
+
* shared system prompt must not be able to inject e.g. blog-publishing into a
|
|
3476
|
+
* non-author's session via requiredTools. (Per-user entitlement of the remaining
|
|
3477
|
+
* tools is still the chat pipeline's responsibility - see follow-up note in the
|
|
3478
|
+
* briefcase blueprint; this allowlist is the storage-layer floor.)
|
|
3479
|
+
*/
|
|
3480
|
+
const BRIEFCASE_DISALLOWED_TOOLS = [
|
|
3481
|
+
"blog_publish",
|
|
3482
|
+
"blog_edit",
|
|
3483
|
+
"blog_draft"
|
|
3484
|
+
];
|
|
3485
|
+
const BriefcaseRequiredToolsSchema = z$1.array(b4mLLMTools.refine((t) => !BRIEFCASE_DISALLOWED_TOOLS.includes(t), "This tool is not permitted in a briefcase prompt")).max(16);
|
|
3486
|
+
z$1.string().regex(/^[a-f0-9]{24}$/i, "Invalid prompt id");
|
|
3487
|
+
/** Shared cap for every free-text prompt body a caller can supply. Exported so the
|
|
3488
|
+
* caller-prompt caps on the chat request, the invoke params and the CLI tool import it
|
|
3489
|
+
* rather than each restating the literal. */
|
|
3490
|
+
const PROMPT_TEXT_MAX = 16e3;
|
|
3491
|
+
const TAGS_MAX = 20;
|
|
3492
|
+
z$1.object({
|
|
3493
|
+
type: z$1.string().min(1).max(100),
|
|
3494
|
+
name: z$1.string().min(1).max(200),
|
|
3495
|
+
description: z$1.string().max(500).optional(),
|
|
3496
|
+
promptText: z$1.string().min(1).max(PROMPT_TEXT_MAX),
|
|
3497
|
+
tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
|
|
3498
|
+
executionMode: AuthorableExecutionModeSchema.optional(),
|
|
3499
|
+
requiredTools: BriefcaseRequiredToolsSchema.optional()
|
|
3500
|
+
}).partial();
|
|
3501
|
+
/**
|
|
3502
|
+
* One catalog sub-query. Exactly one selector is used, in precedence order:
|
|
3503
|
+
* `personal` (resolved to the caller server-side) > `tags` > `type`.
|
|
3504
|
+
*/
|
|
3505
|
+
const PromptBatchQuerySchema = z$1.object({
|
|
3506
|
+
key: z$1.string().min(1).max(100),
|
|
3507
|
+
tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
|
|
3508
|
+
type: z$1.string().max(100).optional(),
|
|
3509
|
+
personal: z$1.boolean().optional()
|
|
3510
|
+
});
|
|
3511
|
+
z$1.object({ queries: z$1.array(PromptBatchQuerySchema).min(1).max(32).refine((qs) => new Set(qs.map((q) => q.key)).size === qs.length, { message: "Batch query keys must be unique" }) });
|
|
3104
3512
|
z$1.object({
|
|
3105
3513
|
sessionId: z$1.string().nullish(),
|
|
3106
3514
|
message: z$1.string(),
|
|
@@ -3116,7 +3524,7 @@ z$1.object({
|
|
|
3116
3524
|
wait: z$1.boolean().prefault(false),
|
|
3117
3525
|
enableTools: z$1.boolean().prefault(false),
|
|
3118
3526
|
toolMode: z$1.enum(["fast", "smart"]).optional(),
|
|
3119
|
-
tools: z$1.array(z$1.string()).optional(),
|
|
3527
|
+
tools: z$1.array(z$1.string()).optional().describe("Explicit tool ids to offer the model. A non-empty array enables tools on its own; no companion `toolMode` or `enableTools` is required. Merged with the auto-selected set under `toolMode: \"smart\"`, and ignored under `toolMode: \"fast\"`. This list ADDS to what is offered rather than restricting it - the server still offers tools of its own (for example knowledge retrieval when the session has reachable documents). Unrecognized ids are dropped rather than rejecting the request; the response reports the surviving set as `tools.effectiveTools` and the ids that are not tools in this deployment as `tools.unrecognizedTools`, since no endpoint enumerates the valid ids. `unrecognizedTools` is reported under every `toolMode` - it describes the ids, not what the mode did with them - so under `toolMode: \"fast\"` an id can be absent from `effectiveTools` (the mode discarded it) without being unrecognized. Both reported lists are deduplicated, and `unrecognizedTools` names at most the first 10 distinct ids."),
|
|
3120
3528
|
enableQuestMaster: z$1.boolean().optional(),
|
|
3121
3529
|
enableMementos: z$1.boolean().optional(),
|
|
3122
3530
|
enableAgents: z$1.boolean().optional(),
|
|
@@ -3125,8 +3533,10 @@ z$1.object({
|
|
|
3125
3533
|
"grounded",
|
|
3126
3534
|
"surface"
|
|
3127
3535
|
]).optional(),
|
|
3536
|
+
skip_auto_offers: z$1.boolean().optional().describe("Suppress tools the server would otherwise attach on its own for this session (the knowledge-base search offer, in-app view navigation, blog drafting/editing/publishing, and skill invocation). Tools you request explicitly are unaffected. One system-prompt block goes with them: withholding in-app view navigation also drops the view-registry block that exists only to describe it. No other prompt content changes. This does not switch off retrieval: a session with forced knowledge retrieval still retrieves, and documents already attached to the session are still placed in the prompt directly. Any promptMode suppresses these too, so false has no effect alongside one."),
|
|
3128
3537
|
includePromptDetails: z$1.boolean().optional(),
|
|
3129
|
-
includeSystemPrompt: z$1.boolean().optional()
|
|
3538
|
+
includeSystemPrompt: z$1.boolean().optional(),
|
|
3539
|
+
systemPrompt: z$1.string().max(PROMPT_TEXT_MAX).optional().describe("System-prompt text for this request only, never persisted. Rendered as a defended block appended after every other system-prompt source, with prose instructing the model to defer to organization, session and data-lake guidance. Over the cap is a 422, never truncated.")
|
|
3130
3540
|
});
|
|
3131
3541
|
z$1.object({
|
|
3132
3542
|
id: z$1.string(),
|
|
@@ -3135,6 +3545,12 @@ z$1.object({
|
|
|
3135
3545
|
timestamp: z$1.string(),
|
|
3136
3546
|
model: z$1.string(),
|
|
3137
3547
|
message: z$1.string().optional(),
|
|
3548
|
+
tools: z$1.object({
|
|
3549
|
+
toolMode: z$1.enum(["fast", "smart"]).optional(),
|
|
3550
|
+
autoSelectedTools: z$1.array(z$1.string()).optional(),
|
|
3551
|
+
effectiveTools: z$1.array(z$1.string()),
|
|
3552
|
+
unrecognizedTools: z$1.array(z$1.string()).optional()
|
|
3553
|
+
}).optional(),
|
|
3138
3554
|
tracking_info: z$1.object({
|
|
3139
3555
|
quest_id: z$1.string(),
|
|
3140
3556
|
check_status_url: z$1.string(),
|
|
@@ -3357,13 +3773,93 @@ const AGENT_EXECUTION_STATUSES = [
|
|
|
3357
3773
|
"failed",
|
|
3358
3774
|
"aborted"
|
|
3359
3775
|
];
|
|
3776
|
+
/**
|
|
3777
|
+
* Operator allow-list for the client-authored Mongo filter carried on a `subscribe_query` frame.
|
|
3778
|
+
*
|
|
3779
|
+
* The WS data-subscribe handler forwards that filter to `Model.find` and persists it on the
|
|
3780
|
+
* QuerySubscription the separate subscriber-fanout service replays, so an unconstrained
|
|
3781
|
+
* `$`-prefixed key lets any authenticated socket ask the database to run server-side JavaScript
|
|
3782
|
+
* (`$where`, `$expr` + `$function`) or an unbounded `$regex` - either of which pins a pooled
|
|
3783
|
+
* connection for as long as it runs.
|
|
3784
|
+
*
|
|
3785
|
+
* Live subscriptions only ever need equality, membership and range matching, so anything outside
|
|
3786
|
+
* this list is REFUSED rather than stripped: silently dropping an operator would quietly widen the
|
|
3787
|
+
* document set the caller ends up subscribed to.
|
|
3788
|
+
*
|
|
3789
|
+
* The scan is deliberately structural rather than a model of Mongo's grammar - it flags any
|
|
3790
|
+
* `$`-prefixed key anywhere in the tree that isn't allow-listed. It can therefore accept an
|
|
3791
|
+
* allow-listed operator in a position Mongo would treat as a literal sub-document field name, but
|
|
3792
|
+
* that is inert; what matters is that no disallowed operator can reach the driver.
|
|
3793
|
+
*/
|
|
3794
|
+
/** Combinators whose operands are themselves filters. */
|
|
3795
|
+
const SUBSCRIPTION_FILTER_LOGICAL_OPERATORS = [
|
|
3796
|
+
"$and",
|
|
3797
|
+
"$or",
|
|
3798
|
+
"$nor",
|
|
3799
|
+
"$not"
|
|
3800
|
+
];
|
|
3801
|
+
/** Value-level operators a subscription filter may use. */
|
|
3802
|
+
const SUBSCRIPTION_FILTER_VALUE_OPERATORS = [
|
|
3803
|
+
"$eq",
|
|
3804
|
+
"$ne",
|
|
3805
|
+
"$gt",
|
|
3806
|
+
"$gte",
|
|
3807
|
+
"$lt",
|
|
3808
|
+
"$lte",
|
|
3809
|
+
"$in",
|
|
3810
|
+
"$nin",
|
|
3811
|
+
"$exists",
|
|
3812
|
+
"$type",
|
|
3813
|
+
"$size",
|
|
3814
|
+
"$all",
|
|
3815
|
+
"$elemMatch"
|
|
3816
|
+
];
|
|
3817
|
+
const ALLOWED_OPERATORS = /* @__PURE__ */ new Set([...SUBSCRIPTION_FILTER_LOGICAL_OPERATORS, ...SUBSCRIPTION_FILTER_VALUE_OPERATORS]);
|
|
3818
|
+
/**
|
|
3819
|
+
* Returns a dotted path for every part of `filter` a subscription may not send: a disallowed
|
|
3820
|
+
* `$`-prefixed key, a `RegExp` operand, or nesting past {@link SUBSCRIPTION_FILTER_MAX_DEPTH}.
|
|
3821
|
+
* An empty array means the filter is safe to forward to the database.
|
|
3822
|
+
*/
|
|
3823
|
+
function findDisallowedSubscriptionFilterKeys(filter) {
|
|
3824
|
+
const violations = [];
|
|
3825
|
+
const walk = (node, path, depth) => {
|
|
3826
|
+
if (depth > 12) {
|
|
3827
|
+
violations.push(`${path} (nested deeper than 12)`);
|
|
3828
|
+
return;
|
|
3829
|
+
}
|
|
3830
|
+
if (node instanceof RegExp) {
|
|
3831
|
+
violations.push(`${path} (regular expression)`);
|
|
3832
|
+
return;
|
|
3833
|
+
}
|
|
3834
|
+
if (Array.isArray(node)) {
|
|
3835
|
+
node.forEach((entry, i) => walk(entry, `${path}[${i}]`, depth + 1));
|
|
3836
|
+
return;
|
|
3837
|
+
}
|
|
3838
|
+
if (node === null || typeof node !== "object") return;
|
|
3839
|
+
for (const [key, value] of Object.entries(node)) {
|
|
3840
|
+
const childPath = path ? `${path}.${key}` : key;
|
|
3841
|
+
if (key.startsWith("$") && !ALLOWED_OPERATORS.has(key)) {
|
|
3842
|
+
violations.push(childPath);
|
|
3843
|
+
continue;
|
|
3844
|
+
}
|
|
3845
|
+
walk(value, childPath, depth + 1);
|
|
3846
|
+
}
|
|
3847
|
+
};
|
|
3848
|
+
walk(filter, "", 0);
|
|
3849
|
+
return violations;
|
|
3850
|
+
}
|
|
3360
3851
|
const DataSubscribeRequestAction = z$1.object({
|
|
3361
3852
|
action: z$1.literal("subscribe_query"),
|
|
3362
3853
|
accessToken: z$1.string().optional(),
|
|
3363
3854
|
subscriptionId: z$1.string(),
|
|
3364
3855
|
collectionName: z$1.string(),
|
|
3365
|
-
query: z$1.looseObject({}),
|
|
3366
|
-
|
|
3856
|
+
query: z$1.looseObject({}).superRefine((filter, ctx) => {
|
|
3857
|
+
for (const key of findDisallowedSubscriptionFilterKeys(filter)) ctx.addIssue({
|
|
3858
|
+
code: "custom",
|
|
3859
|
+
message: `Disallowed subscription filter: ${key}`
|
|
3860
|
+
});
|
|
3861
|
+
}),
|
|
3862
|
+
fields: z$1.record(z$1.string(), z$1.union([z$1.boolean(), z$1.number()])),
|
|
3367
3863
|
fetchInitialData: z$1.boolean().prefault(true).optional(),
|
|
3368
3864
|
clientId: z$1.string().optional()
|
|
3369
3865
|
});
|
|
@@ -3450,6 +3946,17 @@ const DataSubscriptionUpdateAction = z$1.object({
|
|
|
3450
3946
|
id: z$1.string()
|
|
3451
3947
|
})
|
|
3452
3948
|
});
|
|
3949
|
+
/**
|
|
3950
|
+
* Server -> Client: a `subscribe_query` frame was refused (the operator allow-list, `fields`
|
|
3951
|
+
* validation) or its initial fetch was aborted (`maxTimeMS`). Neither failure throws an
|
|
3952
|
+
* UnauthorizedError/JsonWebTokenError, so withWebSocketContext's status code never reaches the
|
|
3953
|
+
* client as a frame - this is the only signal the caller gets that the subscription never took.
|
|
3954
|
+
*/
|
|
3955
|
+
const DataSubscribeErrorAction = z$1.object({
|
|
3956
|
+
action: z$1.literal("data_subscribe_error"),
|
|
3957
|
+
subscriptionId: z$1.string(),
|
|
3958
|
+
error: z$1.string()
|
|
3959
|
+
});
|
|
3453
3960
|
const LLMStatusUpdateAction = z$1.object({
|
|
3454
3961
|
action: z$1.literal("llm_status_update"),
|
|
3455
3962
|
status: z$1.string().nullable(),
|
|
@@ -3532,6 +4039,7 @@ const QuestExportProgressAction = z$1.object({
|
|
|
3532
4039
|
downloadUrl: z$1.string().optional(),
|
|
3533
4040
|
filename: z$1.string().optional(),
|
|
3534
4041
|
errorMessage: z$1.string().optional(),
|
|
4042
|
+
droppedQuestCount: z$1.number().optional(),
|
|
3535
4043
|
clientId: z$1.string().optional()
|
|
3536
4044
|
});
|
|
3537
4045
|
const SpiderProgressUpdateAction = z$1.object({
|
|
@@ -4682,6 +5190,7 @@ const OptiHashiRunUpdatedAction = z$1.object({
|
|
|
4682
5190
|
});
|
|
4683
5191
|
z$1.discriminatedUnion("action", [
|
|
4684
5192
|
DataSubscriptionUpdateAction,
|
|
5193
|
+
DataSubscribeErrorAction,
|
|
4685
5194
|
InboxRefetchAction,
|
|
4686
5195
|
LLMStatusUpdateAction,
|
|
4687
5196
|
InvitesRefetchAction,
|
|
@@ -4739,6 +5248,126 @@ z$1.discriminatedUnion("action", [
|
|
|
4739
5248
|
PermissionRequestAction,
|
|
4740
5249
|
ReconnectResultAction
|
|
4741
5250
|
]);
|
|
5251
|
+
z$1.object({
|
|
5252
|
+
session_id: z$1.string().min(1),
|
|
5253
|
+
message: z$1.string().min(1),
|
|
5254
|
+
/** Falls back to the deployment's default chat model when omitted. */
|
|
5255
|
+
model: z$1.string().optional(),
|
|
5256
|
+
/**
|
|
5257
|
+
* Run as a specific persisted agent. Omit to let the executor pick the profile for
|
|
5258
|
+
* the session: a session on a dedicated surface gets that surface's own profile,
|
|
5259
|
+
* otherwise a synthetic one built from admin orchestration defaults. Omitting this
|
|
5260
|
+
* is what reproduces the product UI's Agent Mode toggle.
|
|
5261
|
+
*/
|
|
5262
|
+
agent_id: z$1.string().optional(),
|
|
5263
|
+
/**
|
|
5264
|
+
* Bill this run to an organization's credit pool. The caller must belong to it; a
|
|
5265
|
+
* non-member gets 404. Omit to bill the caller personally.
|
|
5266
|
+
*/
|
|
5267
|
+
organization_id: z$1.string().optional(),
|
|
5268
|
+
/**
|
|
5269
|
+
* Tool-id allowlist for the run. Omit to use the resolved profile's own list.
|
|
5270
|
+
*
|
|
5271
|
+
* Doubles as pre-approval: REST runs have no interactive client to answer a
|
|
5272
|
+
* permission prompt, so tools named here are treated as approved. A run that calls
|
|
5273
|
+
* an approval-gated tool NOT named here fails with that tool named in `error`.
|
|
5274
|
+
*/
|
|
5275
|
+
tools: z$1.array(z$1.string()).optional(),
|
|
5276
|
+
/**
|
|
5277
|
+
* Hard ceiling on ReAct iterations. Each one is a full LLM round-trip, so the cap is
|
|
5278
|
+
* bounded at 100 regardless of what the profile would allow.
|
|
5279
|
+
*/
|
|
5280
|
+
max_iterations: z$1.number().int().positive().max(100).optional(),
|
|
5281
|
+
temperature: z$1.number().min(0).max(2).optional(),
|
|
5282
|
+
max_tokens: z$1.number().int().positive().optional(),
|
|
5283
|
+
thinking: z$1.object({
|
|
5284
|
+
enabled: z$1.boolean(),
|
|
5285
|
+
/** Bounded at 32000: Anthropic rejects rather than clamps an oversized budget. */
|
|
5286
|
+
budget_tokens: z$1.number().int().positive().max(32e3).optional()
|
|
5287
|
+
}).optional(),
|
|
5288
|
+
/** Per-message file attachments (fabFile ids), materialized into the first iteration. */
|
|
5289
|
+
file_ids: z$1.array(z$1.string()).optional(),
|
|
5290
|
+
/** Workbench-level file ids for the session, forwarded as a dispatch-time snapshot. */
|
|
5291
|
+
session_file_ids: z$1.array(z$1.string()).optional(),
|
|
5292
|
+
enable_mementos: z$1.boolean().optional(),
|
|
5293
|
+
enable_lattice: z$1.boolean().optional(),
|
|
5294
|
+
/**
|
|
5295
|
+
* Opt out of the artifact-emission prompt and artifact persistence for this run.
|
|
5296
|
+
*
|
|
5297
|
+
* ANDed with the deployment's admin `EnableArtifacts` setting, so this can only ever
|
|
5298
|
+
* withhold artifacts, never force them on. Omitting it means "no preference" and
|
|
5299
|
+
* leaves the admin setting as the only gate; only an explicit `false` opts out.
|
|
5300
|
+
* Inherited by any subagent this run dispatches, so a delegating agent cannot route
|
|
5301
|
+
* around the opt-out.
|
|
5302
|
+
*
|
|
5303
|
+
* Worth setting on a REST run: the emission prompt costs roughly 2.8k tokens per
|
|
5304
|
+
* iteration, and nothing on this transport renders an artifact back to a human.
|
|
5305
|
+
*/
|
|
5306
|
+
enable_artifacts: z$1.boolean().optional()
|
|
5307
|
+
});
|
|
5308
|
+
z$1.object({
|
|
5309
|
+
id: z$1.string(),
|
|
5310
|
+
status: z$1.literal("pending"),
|
|
5311
|
+
session_id: z$1.string(),
|
|
5312
|
+
model: z$1.string(),
|
|
5313
|
+
timestamp: z$1.string(),
|
|
5314
|
+
tracking_info: z$1.object({
|
|
5315
|
+
execution_id: z$1.string(),
|
|
5316
|
+
/**
|
|
5317
|
+
* The chat-history Quest holding the prompt, which gains the reply when the run
|
|
5318
|
+
* completes. Absent when that best-effort write failed; the run still proceeds.
|
|
5319
|
+
*/
|
|
5320
|
+
quest_id: z$1.string().optional(),
|
|
5321
|
+
poll_url: z$1.string()
|
|
5322
|
+
})
|
|
5323
|
+
});
|
|
5324
|
+
/** One step of a published reasoning trace - a public projection of `IAgentStep`. */
|
|
5325
|
+
const AgentExecutionStepSchema = z$1.object({
|
|
5326
|
+
type: z$1.enum([
|
|
5327
|
+
"thought",
|
|
5328
|
+
"action",
|
|
5329
|
+
"observation",
|
|
5330
|
+
"final_answer"
|
|
5331
|
+
]),
|
|
5332
|
+
content: z$1.string(),
|
|
5333
|
+
/** 0-indexed iteration. Absent on traces checkpointed before the field existed. */
|
|
5334
|
+
iteration: z$1.number().int().nonnegative().optional(),
|
|
5335
|
+
/** Set on `action` steps: the tool the agent invoked. */
|
|
5336
|
+
tool_name: z$1.string().optional()
|
|
5337
|
+
});
|
|
5338
|
+
z$1.object({
|
|
5339
|
+
id: z$1.string(),
|
|
5340
|
+
status: z$1.enum([
|
|
5341
|
+
"pending",
|
|
5342
|
+
"running",
|
|
5343
|
+
"continuing",
|
|
5344
|
+
"awaiting_permission",
|
|
5345
|
+
"awaiting_subagent",
|
|
5346
|
+
"awaiting_dag_children",
|
|
5347
|
+
"paused",
|
|
5348
|
+
"completed",
|
|
5349
|
+
"failed",
|
|
5350
|
+
"aborted"
|
|
5351
|
+
]),
|
|
5352
|
+
session_id: z$1.string().nullable(),
|
|
5353
|
+
answer: z$1.string().nullable(),
|
|
5354
|
+
/**
|
|
5355
|
+
* Why the run ended without an answer. Set only on `failed`; null otherwise.
|
|
5356
|
+
* Without this a caller polling a terminal run sees `failed` + a null answer and
|
|
5357
|
+
* cannot tell an approval-gated tool from a model error from a timeout.
|
|
5358
|
+
*
|
|
5359
|
+
* Approval-gate failures name the offending tool, since that is what the caller acts
|
|
5360
|
+
* on. Everything else is reduced to a coarse category (billing, rate limit, timeout,
|
|
5361
|
+
* auth) or a generic message: the stored reason is a raw internal exception, and
|
|
5362
|
+
* those carry infrastructure identifiers. Full detail stays in the server logs.
|
|
5363
|
+
*/
|
|
5364
|
+
error: z$1.string().nullable(),
|
|
5365
|
+
steps: z$1.array(AgentExecutionStepSchema),
|
|
5366
|
+
total_iterations: z$1.number().nullable(),
|
|
5367
|
+
created_at: z$1.string(),
|
|
5368
|
+
updated_at: z$1.string()
|
|
5369
|
+
});
|
|
5370
|
+
z$1.object({ id: z$1.string().min(1) });
|
|
4742
5371
|
/**
|
|
4743
5372
|
* Tool schema matching ICompletionOptionTools.toolSchema. The Zod surface only
|
|
4744
5373
|
* covers wire-format fields (toolFn is server-side). Replaces the historical
|
|
@@ -5813,43 +6442,129 @@ z$2.object({
|
|
|
5813
6442
|
*/
|
|
5814
6443
|
const OVERSIZED_PASSAGE_TOKEN_THRESHOLD = 1500;
|
|
5815
6444
|
/**
|
|
5816
|
-
*
|
|
5817
|
-
*
|
|
5818
|
-
*
|
|
6445
|
+
* Why a file's chunk/vector pipeline is STALLED, stored in `FabFile.chunkStallReason`.
|
|
6446
|
+
*
|
|
6447
|
+
* - `vectorizePaused`: the data-lake convergence kill switch abandoned a vectorize (#1676). The
|
|
6448
|
+
* file keeps its chunks but has no vectors, so it is unsearchable until re-indexed.
|
|
6449
|
+
* - `rechunkPaused`: the OTHER half of the same switch dropped a re-chunk before it ran
|
|
6450
|
+
* (#1676/#1681). The damage is worse - the producer resets a wave's chunk state BEFORE the
|
|
6451
|
+
* messages are handled, so a file halted here has NO chunks at all.
|
|
6452
|
+
* - `unchunkedPaused`: the same chunk half, on a file that never had passages to lose. The rescue
|
|
6453
|
+
* sweep selects on `chunkCount: 0` and enqueues without resetting anything, so a file it routes
|
|
6454
|
+
* into the halt branch arrives already empty. The halted STATE is identical to `rechunkPaused`
|
|
6455
|
+
* and every reader keys on both (CHUNKLESS_STALL_REASONS) - the split exists so the owner is not
|
|
6456
|
+
* told that passages were removed which never existed.
|
|
6457
|
+
*
|
|
6458
|
+
* None auto-resumes; each needs a reprocess or a lifted switch.
|
|
5819
6459
|
*
|
|
5820
|
-
*
|
|
5821
|
-
*
|
|
5822
|
-
*
|
|
5823
|
-
*
|
|
6460
|
+
* Without a marker the state is misread by every surface at once, which is the failure the field
|
|
6461
|
+
* exists to prevent: `chunkCount: 0` with `error: null` reads as an image or a pending upload, so
|
|
6462
|
+
* health drops it from the denominator, convergence grades it `conformant` (its stale stamp still
|
|
6463
|
+
* matches), and search does not withhold it because it is not "in flight". The file's passages are
|
|
6464
|
+
* simply gone and nothing REPORTS it. The rescue sweep is the one exception and deliberately so: an
|
|
6465
|
+
* unmarked file matches its filter and gets re-chunked, which is repair rather than reporting. That
|
|
6466
|
+
* is why the sweep excludes a stalled file only while the switch is ON - see
|
|
6467
|
+
* buildFabFileChunkScanFilter.
|
|
6468
|
+
*
|
|
6469
|
+
* A dedicated field rather than prose in `FabFile.notes` (#2016): `notes` is the USER's note, and
|
|
6470
|
+
* while the markers lived there every writer of the field clobbered the others - a "Rebuild
|
|
6471
|
+
* passages" wave silently deleted whatever the owner had typed.
|
|
6472
|
+
*
|
|
6473
|
+
* Lives here rather than beside its writers (apps/client's chunk and vectorize handlers) because it
|
|
6474
|
+
* is a cross-layer contract: the queue handlers write it and b4m-core's evaluators
|
|
6475
|
+
* (constants/lakeHealth.ts, constants/lakeConvergence.ts, dataLakeService/retrievalUnavailable.ts)
|
|
6476
|
+
* read it to tell a permanently-stalled file from one still in flight. b4m-core cannot import from
|
|
6477
|
+
* apps/client, so a copy there would have to drift silently.
|
|
6478
|
+
*/
|
|
6479
|
+
const CHUNK_STALL_REASONS = [
|
|
6480
|
+
"vectorizePaused",
|
|
6481
|
+
"rechunkPaused",
|
|
6482
|
+
"unchunkedPaused"
|
|
6483
|
+
];
|
|
6484
|
+
/**
|
|
6485
|
+
* Whether a file is stalled by the convergence kill switch, by any arm. THE predicate every
|
|
6486
|
+
* reader uses, so adding a stall reason reaches health, convergence and retrieval without separate
|
|
6487
|
+
* comparisons drifting apart. Also the in-memory mirror of a Mongo
|
|
6488
|
+
* `chunkStallReason: { $in: [...CHUNK_STALL_REASONS] }`.
|
|
5824
6489
|
*/
|
|
5825
|
-
|
|
6490
|
+
function isChunkStalled(reason) {
|
|
6491
|
+
return CHUNK_STALL_REASONS.includes(reason);
|
|
6492
|
+
}
|
|
5826
6493
|
/**
|
|
5827
|
-
*
|
|
5828
|
-
*
|
|
5829
|
-
*
|
|
5830
|
-
*
|
|
6494
|
+
* Which reasons leave the file with NO passages, as opposed to passages with no vectors. A `Record`
|
|
6495
|
+
* over every reason rather than a hand-written subset array: a new stall reason then cannot compile
|
|
6496
|
+
* until it is classified, where a member missing from a literal array would just make a health count
|
|
6497
|
+
* silently wrong.
|
|
6498
|
+
*/
|
|
6499
|
+
const STALL_LEAVES_NO_PASSAGES = {
|
|
6500
|
+
vectorizePaused: false,
|
|
6501
|
+
rechunkPaused: true,
|
|
6502
|
+
unchunkedPaused: true
|
|
6503
|
+
};
|
|
6504
|
+
CHUNK_STALL_REASONS.filter((reason) => STALL_LEAVES_NO_PASSAGES[reason]);
|
|
6505
|
+
/**
|
|
6506
|
+
* Owner-facing prose for a stall reason, and the ONLY place it is worded.
|
|
5831
6507
|
*
|
|
5832
|
-
*
|
|
5833
|
-
*
|
|
5834
|
-
*
|
|
5835
|
-
*
|
|
5836
|
-
|
|
6508
|
+
* `vectorizePaused` and `rechunkPaused` are the exact strings those markers used while they lived in
|
|
6509
|
+
* `notes`, which is also what the #2016 migration matches on to derive the field for existing rows -
|
|
6510
|
+
* do not reword either without updating it. `unchunkedPaused` postdates that migration and was never
|
|
6511
|
+
* written to `notes`, so its wording is free to change: nothing matches on it.
|
|
6512
|
+
*/
|
|
6513
|
+
const CHUNK_STALL_NOTICES = {
|
|
6514
|
+
vectorizePaused: "Indexing paused by the data-lake convergence kill switch - reprocess to complete.",
|
|
6515
|
+
rechunkPaused: "Re-chunking paused by the data-lake convergence kill switch - its passages were removed and are rebuilt when convergence resumes.",
|
|
6516
|
+
unchunkedPaused: "Chunking paused by the data-lake convergence kill switch - this file has no passages yet and they are built when convergence resumes."
|
|
6517
|
+
};
|
|
6518
|
+
CHUNK_STALL_NOTICES.vectorizePaused;
|
|
6519
|
+
CHUNK_STALL_NOTICES.rechunkPaused;
|
|
6520
|
+
/**
|
|
6521
|
+
* TRANSITIONAL, and the ONE stall predicate every RETRIEVAL path must use until #2016's migration
|
|
6522
|
+
* has run in every environment. Reads the new field, then falls back to the legacy prose that the
|
|
6523
|
+
* pre-migration rows still carry in `notes`.
|
|
6524
|
+
*
|
|
6525
|
+
* It exists for the FORWARD window only: `migratorInvocation` is a `dependsOn` of the web stack
|
|
6526
|
+
* only (infra/web.ts); the queue stack has none, so the executor can serve forced retrieval and
|
|
6527
|
+
* `knowledge_base_search` while rows still carry the marker in `notes` and no `chunkStallReason`. A
|
|
6528
|
+
* row stalled by the chunk arm then reads as a plain unindexed file: `isRetrievalExcluded` drops it
|
|
6529
|
+
* upstream of the withhold on a vectorizedOnly lake, and `partitionByIndexAvailability` calls it
|
|
6530
|
+
* servable everywhere else. The turn answers around a passage-less file and reports FULL coverage -
|
|
6531
|
+
* the silent degradation this whole path exists to prevent.
|
|
6532
|
+
*
|
|
6533
|
+
* A code ROLLBACK is the mirror image and this arm CANNOT cover it: the rows are already migrated
|
|
6534
|
+
* (`chunkStallReason` set, `notes` unset) and the code restored is pre-#2016, which does not contain
|
|
6535
|
+
* this function. Nothing reverts the data on its own either - `migratorInvocation` only ever runs
|
|
6536
|
+
* `up` and `migrate down` is a manual CLI step - so `migrate down` is a REQUIRED step of any
|
|
6537
|
+
* rollback past #2016, not an optional tidy-up. What this arm does buy is that `down()` is safe to
|
|
6538
|
+
* run FIRST: whichever stack is still new keeps honoring the prose it restores, so a staggered
|
|
6539
|
+
* rollback has no window where a restored marker is invisible. `down()` is a PARTIAL restore
|
|
6540
|
+
* though - it skips a row whose owner typed a note after `up()`, and that row grades as unstalled
|
|
6541
|
+
* on both stacks once the field is dropped. See its own comment.
|
|
6542
|
+
*
|
|
6543
|
+
* Deliberately NOT used by the grading/health/UI readers: they are gated behind the web stack, and
|
|
6544
|
+
* a legacy row there renders the notice line AND the identical text as the owner's note.
|
|
5837
6545
|
*
|
|
5838
|
-
*
|
|
5839
|
-
*
|
|
6546
|
+
* Mirrored in Mongo by `buildFabFileSearchQuery`'s `vectorizedOnly` exemption. Delete the legacy arm
|
|
6547
|
+
* from both together, one release after the migration has landed everywhere.
|
|
6548
|
+
*
|
|
6549
|
+
* Pinned to the two reasons the migration backfilled rather than every notice: `unchunkedPaused`
|
|
6550
|
+
* postdates it, so no row carries its prose, and including it would read an owner who happens to type
|
|
6551
|
+
* that sentence into `notes` as stalled.
|
|
5840
6552
|
*/
|
|
5841
|
-
const
|
|
6553
|
+
const LEGACY_CHUNK_STALL_NOTES = [CHUNK_STALL_NOTICES.vectorizePaused, CHUNK_STALL_NOTICES.rechunkPaused];
|
|
6554
|
+
function isChunkStalledFile(file) {
|
|
6555
|
+
return isChunkStalled(file.chunkStallReason) || LEGACY_CHUNK_STALL_NOTES.includes(file.notes ?? "");
|
|
6556
|
+
}
|
|
5842
6557
|
/**
|
|
5843
6558
|
* `FabFile.chunkRebuildRequestedAt`: stamped by `resetChunkStateByIds` in the SAME write that
|
|
5844
6559
|
* clears a file's chunk rollups, so "this file's passages are being rebuilt" can never be lost the
|
|
5845
6560
|
* way the pair of steps that creates the state can be. The reset and the queue send are two
|
|
5846
6561
|
* operations - kill the producer between them, or lose the consumer's marker write, and the file
|
|
5847
|
-
* sits at `chunkCount: 0` with `error: null` and
|
|
6562
|
+
* sits at `chunkCount: 0` with `error: null` and no stall reason, a shape indistinguishable from an
|
|
5848
6563
|
* image or a still-uploading row. It then drops out of lake health's denominator, out of the
|
|
5849
6564
|
* convergence plan and out of the retrieval withhold at the same moment: every rollup says its
|
|
5850
6565
|
* passages are gone, and nothing reports it.
|
|
5851
6566
|
*
|
|
5852
|
-
* Deliberately NOT `
|
|
6567
|
+
* Deliberately NOT the `rechunkPaused` stall reason pre-written by the producer, which is the obvious
|
|
5853
6568
|
* fix and the wrong one: that marker means "halted, needs an administrator", so a file awaiting an
|
|
5854
6569
|
* ORDINARY rebuild would read to every reader as permanently paused for the whole rebuild - search
|
|
5855
6570
|
* would tell readers it does not return on its own, health would hard-fail P3, and "Rebuild
|
|
@@ -5861,9 +6576,9 @@ const CONVERGENCE_PAUSED_CHUNK_NOTE = "Re-chunking paused by the data-lake conve
|
|
|
5861
6576
|
* upgrade therefore degrades to mislabelled-but-visible rather than invisible, which is the trade
|
|
5862
6577
|
* this field exists to make - invisibility is the real harm, labelling is secondary.
|
|
5863
6578
|
*
|
|
5864
|
-
* A dedicated field
|
|
5865
|
-
*
|
|
5866
|
-
*
|
|
6579
|
+
* A dedicated field on purpose, and the precedent #2016 followed for the other two machine-written
|
|
6580
|
+
* facts: while they all shared `notes` every writer of that field clobbered the others, including
|
|
6581
|
+
* the user's own note.
|
|
5867
6582
|
*
|
|
5868
6583
|
* Cleared by `commitFabFileChunks` (the rebuild landed) and by the chunk handler's pause write (the
|
|
5869
6584
|
* rebuild was halted instead). A file carrying `error` is settled regardless - see
|
|
@@ -5872,20 +6587,6 @@ const CONVERGENCE_PAUSED_CHUNK_NOTE = "Re-chunking paused by the data-lake conve
|
|
|
5872
6587
|
function isChunkRebuildPending(requestedAt) {
|
|
5873
6588
|
return requestedAt !== null && requestedAt !== void 0 && requestedAt !== "";
|
|
5874
6589
|
}
|
|
5875
|
-
/**
|
|
5876
|
-
* Whether a file's `notes` marks it as stalled by the convergence kill switch, by either arm.
|
|
5877
|
-
* THE predicate every reader uses, so adding a third stall marker reaches health, convergence and
|
|
5878
|
-
* retrieval without three separate string comparisons drifting apart.
|
|
5879
|
-
*/
|
|
5880
|
-
function isConvergencePausedNote(notes) {
|
|
5881
|
-
return CONVERGENCE_PAUSED_NOTES.includes(notes);
|
|
5882
|
-
}
|
|
5883
|
-
/**
|
|
5884
|
-
* Datastore mirror of `isConvergencePausedNote`, for a Mongo `notes: { $in: [...] }`. Exported so a
|
|
5885
|
-
* query and the in-memory predicate cannot drift: adding a third stall marker to this array reaches
|
|
5886
|
-
* both. Declared after the two constants it names so the function above can close over it.
|
|
5887
|
-
*/
|
|
5888
|
-
const CONVERGENCE_PAUSED_NOTES = [CONVERGENCE_PAUSED_NOTE, CONVERGENCE_PAUSED_CHUNK_NOTE];
|
|
5889
6590
|
/** Ceiling so "adjustable" cannot mean "unbounded" in either direction. */
|
|
5890
6591
|
const LAKE_ACCESS_AUDIT_RETENTION_MAX_DAYS = 2555;
|
|
5891
6592
|
/**
|
|
@@ -5917,14 +6618,13 @@ const LAKE_CONFIG_AUDIT_RETENTION_DEFAULT_DAYS = 1095;
|
|
|
5917
6618
|
/** Ceiling so "adjustable" cannot mean "unbounded" in either direction. */
|
|
5918
6619
|
const LAKE_CONFIG_AUDIT_RETENTION_MAX_DAYS = 3650;
|
|
5919
6620
|
/**
|
|
5920
|
-
* Forced-retrieval budget defaults, shared between the admin-settings schema in this
|
|
5921
|
-
* `ChatCompletionFeatures.ts` (which cannot import from `common`'s settings schema
|
|
5922
|
-
* dependency cycle, so the
|
|
6621
|
+
* Forced-retrieval budget and relevance defaults, shared between the admin-settings schema in this
|
|
6622
|
+
* package and `ChatCompletionFeatures.ts` (which cannot import from `common`'s settings schema
|
|
6623
|
+
* without a dependency cycle, so the constants live here instead).
|
|
5923
6624
|
*
|
|
5924
|
-
*
|
|
5925
|
-
*
|
|
5926
|
-
*
|
|
5927
|
-
* stay next to each other, not because it is tunable today.
|
|
6625
|
+
* All three are levers. The char budget is the measured binding constraint on how much of a corpus
|
|
6626
|
+
* reaches the model on every Data-Lake-mode turn; the two floors decide which passages are eligible
|
|
6627
|
+
* to spend it (see `forcedRetrievalRelativeFloorPct` and `forcedRetrievalMinSimilarityPct`).
|
|
5928
6628
|
*/
|
|
5929
6629
|
/** Total characters of retrieved chunk text injected into a forced-retrieval prompt. */
|
|
5930
6630
|
const FORCED_RETRIEVAL_CHAR_BUDGET_DEFAULT = 12e3;
|
|
@@ -5958,12 +6658,23 @@ let OllamaEmbeddingModel = /* @__PURE__ */ function(OllamaEmbeddingModel) {
|
|
|
5958
6658
|
return OllamaEmbeddingModel;
|
|
5959
6659
|
}({});
|
|
5960
6660
|
/**
|
|
5961
|
-
* The default embedding model
|
|
5962
|
-
*
|
|
5963
|
-
*
|
|
5964
|
-
*
|
|
5965
|
-
*
|
|
5966
|
-
*
|
|
6661
|
+
* The default embedding model this deployment advertises - the `defaultEmbeddingModel` admin-setting
|
|
6662
|
+
* default and the query-embedding fallback, so an operator who never opens admin settings still gets
|
|
6663
|
+
* a sensible model rather than an unconfigured one.
|
|
6664
|
+
*
|
|
6665
|
+
* Deliberately answers from the ENVIRONMENT only, and deliberately does NOT try to answer "can this
|
|
6666
|
+
* deployment actually reach that model". It cannot: on a hosted stage an SST secret arrives as a
|
|
6667
|
+
* linked Resource, never as process.env (see apps/client/server/modelDiscovery/adapters.ts), so an
|
|
6668
|
+
* absent OPENAI_API_KEY here means "unknown", not "no key" - it is absent on production too. This
|
|
6669
|
+
* function is also bundled into the browser via settingsMap, where every env read is undefined, so
|
|
6670
|
+
* any answer it gives must be one that is safe as a stage-neutral constant.
|
|
6671
|
+
*
|
|
6672
|
+
* Reachability is therefore decided where the resolved credentials are actually in hand, by
|
|
6673
|
+
* `resolveEmbeddingWithKeylessFallback` (fab-pipeline) - that is what lets a keyless cloud stage
|
|
6674
|
+
* embed on Bedrock without changing what any other stage advertises.
|
|
6675
|
+
*
|
|
6676
|
+
* The one arm that IS env-decidable: a self-host running Ollama has its base URL in real process.env
|
|
6677
|
+
* on both server and (absent) client, and no cloud credential can arrive by any other route.
|
|
5967
6678
|
*/
|
|
5968
6679
|
function defaultEmbeddingModelForEnv() {
|
|
5969
6680
|
const selfHost = process.env.B4M_SELF_HOST === "true";
|
|
@@ -5972,6 +6683,45 @@ function defaultEmbeddingModelForEnv() {
|
|
|
5972
6683
|
if (selfHost && hasOllama && !hasCloudEmbeddingKey) return "qwen3-embedding:0.6b";
|
|
5973
6684
|
return "text-embedding-ada-002";
|
|
5974
6685
|
}
|
|
6686
|
+
/**
|
|
6687
|
+
* True when this deployment can embed with no provider API key at all: a cloud stage reaches
|
|
6688
|
+
* Bedrock through its task/execution role's AWS credentials.
|
|
6689
|
+
*
|
|
6690
|
+
* Requires POSITIVE evidence of an execution role rather than merely "not self-host". A plain
|
|
6691
|
+
* `next dev` session and a CI job both leave B4M_SELF_HOST unset while holding no AWS credentials
|
|
6692
|
+
* at all, so an absence test would send them to the Bedrock SDK for an opaque `CredentialsProvider
|
|
6693
|
+
* Error` in place of the actionable OPENAI_KEY_MISSING_MESSAGE naming the key to set - the same
|
|
6694
|
+
* actionable-to-opaque trade this fallback exists to avoid, just in a different keyless place.
|
|
6695
|
+
*
|
|
6696
|
+
* BOTH runtimes must be covered, and they carry different markers. The Lambdas (vectorize
|
|
6697
|
+
* subscriber, the crons, the Next API routes) get AWS_LAMBDA_FUNCTION_NAME; ChatCompletion is a
|
|
6698
|
+
* Fargate service (infra/chatCompletion.ts) and gets the ECS task-role URI instead. Since
|
|
6699
|
+
* knowledgeBaseSearch runs inside that container, keying on the Lambda marker alone would leave
|
|
6700
|
+
* chat knowledge-base search failing on exactly the keyless stages this fallback is for.
|
|
6701
|
+
*
|
|
6702
|
+
* SST_RESOURCE_App is the third arm and the one this repo can prove: SST sets it on anything it
|
|
6703
|
+
* links, Lambda and Service alike (infra/chatCompletion.ts:129 and infra/agentExecutor.ts:54 both
|
|
6704
|
+
* note that linking alone exposes SST_RESOURCE_*). It covers the Fargate task whether or not the
|
|
6705
|
+
* ECS credential URI is present, and it is absent from a plain `next dev` and from CI, which is
|
|
6706
|
+
* the case that matters. `sst dev` does set it - correctly, since that session runs against real
|
|
6707
|
+
* AWS credentials.
|
|
6708
|
+
*
|
|
6709
|
+
* Self-host is excluded outright because it has no such role - its keyless path is the local
|
|
6710
|
+
* Ollama embedder (`isLocalEmbedderAvailable` in toolAvailability.ts), not Bedrock.
|
|
6711
|
+
*
|
|
6712
|
+
* Answers "is Bedrock reachable here", NOT "should we use it" - a keyed stage is keyless-capable
|
|
6713
|
+
* too, so this must only ever be asked ALONGSIDE a resolved credential table that came back empty.
|
|
6714
|
+
* `resolveEmbeddingWithKeylessFallback` is where the two questions are paired for callers free to
|
|
6715
|
+
* choose the model, and is what such a caller should use instead of asking this directly. The
|
|
6716
|
+
* direct callers are the ones that additionally need the answer BEFORE resolving, to decide
|
|
6717
|
+
* policy: toolAvailability reports whether embedding-backed tools are usable at all, and
|
|
6718
|
+
* data-lakes/semantic-search decides whether a substitution is permitted for this request before
|
|
6719
|
+
* it knows whether one is needed. Both still pair it with the table.
|
|
6720
|
+
*/
|
|
6721
|
+
function hasKeylessCloudEmbedder() {
|
|
6722
|
+
if (process.env.B4M_SELF_HOST === "true") return false;
|
|
6723
|
+
return !!(process.env.AWS_LAMBDA_FUNCTION_NAME || process.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI || process.env.AWS_CONTAINER_CREDENTIALS_FULL_URI || process.env.SST_RESOURCE_App);
|
|
6724
|
+
}
|
|
5975
6725
|
z$1.union([
|
|
5976
6726
|
z$1.enum(OpenAIEmbeddingModel),
|
|
5977
6727
|
z$1.enum(VoyageAIEmbeddingModel),
|
|
@@ -5979,6 +6729,18 @@ z$1.union([
|
|
|
5979
6729
|
z$1.enum(OllamaEmbeddingModel)
|
|
5980
6730
|
]);
|
|
5981
6731
|
/**
|
|
6732
|
+
* The measured per-space floors, rendered for an admin-facing description (e.g. "75 for
|
|
6733
|
+
* text-embedding-ada-002, 35 for text-embedding-3-small").
|
|
6734
|
+
*
|
|
6735
|
+
* Rendered rather than written out in prose because these numbers are expected to move - 35 is
|
|
6736
|
+
* provisional until it is re-derived against a production lake - and a description that restates
|
|
6737
|
+
* the table is a wrong number shown to operators the moment it drifts, with nothing failing.
|
|
6738
|
+
*/
|
|
6739
|
+
const forcedRetrievalFloorsBySpaceSummary = Object.entries({
|
|
6740
|
+
["text-embedding-ada-002"]: 75,
|
|
6741
|
+
["text-embedding-3-small"]: 35
|
|
6742
|
+
}).map(([space, pct]) => `${pct} for ${space}`).join(", ");
|
|
6743
|
+
/**
|
|
5982
6744
|
* Default text for the artifact-emission system prompt. Single source of truth used BOTH as the
|
|
5983
6745
|
* `ArtifactEmissionPrompt` admin setting's default AND as the runtime fallback in
|
|
5984
6746
|
* ChatCompletionProcess - so an unset/empty/cleared DB value reverts to this and never bricks
|
|
@@ -6082,6 +6844,76 @@ const HELP_CENTER_PROMPT = `HELP CENTER: Bike4Mind has a built-in Help Center th
|
|
|
6082
6844
|
*/
|
|
6083
6845
|
const ABSTENTION_PROMPT = `When a request is underspecified or your sources do not cover it, say so and name what is missing. "I do not have enough to answer that" is a correct, high-value answer. Never invent facts about the user, their business, or their data, and never state a specific customer, competitor, deal, or figure as fact - or cite a source for it - unless your sources support it, even when the question assumes it.`;
|
|
6084
6846
|
/**
|
|
6847
|
+
* Default text for the web-search freshness nudge, and the `WebSearchFreshnessPrompt` admin
|
|
6848
|
+
* setting's default.
|
|
6849
|
+
*
|
|
6850
|
+
* Unlike ABSTENTION_PROMPT / ARTIFACT_EMISSION_PROMPT / HELP_CENTER_PROMPT, this setting
|
|
6851
|
+
* distinguishes an absent row from a cleared one. ChatCompletionProcess reads it 2-arg, so an
|
|
6852
|
+
* absent row still falls back to this constant as the setting's registered default, but a cleared
|
|
6853
|
+
* '' is returned verbatim and drops the section rather than reverting. The siblings are read 3-arg
|
|
6854
|
+
* and collapse both cases to the constant. That divergence is deliberate - this section has no
|
|
6855
|
+
* companion boolean, so clearing the field is the only off switch it has. Keep the setting's
|
|
6856
|
+
* description in sync with that if either changes.
|
|
6857
|
+
*
|
|
6858
|
+
* Names no tool but `web_search`: the section is gated on web_search being offered, and web_fetch
|
|
6859
|
+
* is an independent toggle that may well be off.
|
|
6860
|
+
*/
|
|
6861
|
+
const WEB_SEARCH_FRESHNESS_PROMPT = `# WEB SEARCH AND FRESHNESS
|
|
6862
|
+
|
|
6863
|
+
Your training data has a cutoff. The current date is supplied to you in this conversation's system context - treat it as authoritative, and assume anything time-sensitive may have changed since your training.
|
|
6864
|
+
|
|
6865
|
+
Call \`web_search\` BEFORE answering when the answer depends on a fact that changes over time: current prices or rates, product availability or roadmap status, funding, organizational or personnel changes, published benchmarks or performance figures, competitive positioning, or anything the user frames as "current", "latest", "now", or "as of today". When a stale answer would mislead, search instead of answering from memory. When a search surfaces a specific page that matters, or the user names one, read that page directly rather than answering from the snippet.
|
|
6866
|
+
|
|
6867
|
+
You do not need to search for stable knowledge (definitions, mathematics, established theory), or for questions answerable purely from this conversation or from documents already retrieved for you.
|
|
6868
|
+
|
|
6869
|
+
When you report a time-sensitive fact, state what it is as of - the date of the source you used - and say plainly when you could not verify something and are answering from training data instead. Never present an unverified recollection as a current fact.`;
|
|
6870
|
+
/**
|
|
6871
|
+
* Default text for the knowledge-base retrieval nudge, and the `KnowledgeBaseRetrievalPrompt`
|
|
6872
|
+
* admin setting's default.
|
|
6873
|
+
*
|
|
6874
|
+
* The gap this closes: the tool prompt has a when-to-use section for the clock, for web search,
|
|
6875
|
+
* for MCP and for agent delegation, and none for the user's own corpus. The
|
|
6876
|
+
* `search_knowledge_base` description is entirely HOW to search ("Make ONE good search per
|
|
6877
|
+
* distinct topic") and never WHEN, so on the optional path the model decides unaided - and over 30
|
|
6878
|
+
* days of production it reached for the corpus on 20.1% of the turns it was offered on.
|
|
6879
|
+
*
|
|
6880
|
+
* Read 2-arg by ChatCompletionProcess, exactly as WEB_SEARCH_FRESHNESS_PROMPT is and unlike the
|
|
6881
|
+
* 3-arg siblings: an absent row falls back to this constant as the registered default, but a
|
|
6882
|
+
* cleared '' is returned verbatim and drops the section instead of reverting. Deliberate - the
|
|
6883
|
+
* section has no companion boolean, so clearing the field is its only off switch, and that off
|
|
6884
|
+
* switch is what makes it A/B-able without a deploy. Keep the setting's description in sync.
|
|
6885
|
+
*
|
|
6886
|
+
* Names no tool but `search_knowledge_base`, for the same reason the web-search section names no
|
|
6887
|
+
* `web_fetch`: the companion `retrieve_knowledge_content` is paired in at build time but a session
|
|
6888
|
+
* denylist can still strip it (ChatCompletionProcess warns on exactly that case), and instructing
|
|
6889
|
+
* the model to call a tool it was not given makes it emit the call as leaked JSON text.
|
|
6890
|
+
*
|
|
6891
|
+
* The "do not search" paragraph is load-bearing, not padding. A when-to-retrieve nudge without a
|
|
6892
|
+
* don't-retrieve clause buys retrieval on turns that need none - the same failure mode global
|
|
6893
|
+
* forced retrieval already shows on out-of-corpus questions, reached by a different route. Three of
|
|
6894
|
+
* its clauses are load-bearing for a specific co-resident path, not general hedging:
|
|
6895
|
+
* - "from an attached document" - a small attached corpus is INLINED rather than deferred to
|
|
6896
|
+
* retrieval (`shouldDeferCorpusToRetrieval`), and forced retrieval deliberately steps aside on
|
|
6897
|
+
* an attached-files turn (`forcedRetrievalAbstention` emits nothing there). Without this clause
|
|
6898
|
+
* the section tells the model to go searching for content already sitting in its context.
|
|
6899
|
+
* - "already been searched on this turn" - on a forced turn that found nothing,
|
|
6900
|
+
* `forcedRetrievalNoContextPrompt` instructs the model to say the library does not cover the
|
|
6901
|
+
* question. A nudge to search then invites a second identical query - a billed query embedding,
|
|
6902
|
+
* and a chance to talk itself out of a correct abstention.
|
|
6903
|
+
* - the opening scope, "unless its content has been placed in this conversation" - the reason the
|
|
6904
|
+
* first paragraph does not simply claim the documents are invisible, which is false whenever a
|
|
6905
|
+
* corpus was inlined.
|
|
6906
|
+
*/
|
|
6907
|
+
const KNOWLEDGE_BASE_RETRIEVAL_PROMPT = `# KNOWLEDGE BASE
|
|
6908
|
+
|
|
6909
|
+
\`search_knowledge_base\` searches a library of documents the user has made available to you - their own uploads, and any shared or organization library they can reach. You cannot see what a document holds unless its content has been placed in this conversation or you search for it; file names and tags are labels, not content.
|
|
6910
|
+
|
|
6911
|
+
Call \`search_knowledge_base\` BEFORE answering when that library would settle the question: anything about their organization, projects, customers, products, processes or people; a term, name, acronym or identifier that is not general public knowledge; a policy, decision, figure or date specific to them; or a question that assumes context this conversation never gave you. If you are about to answer in general terms a question the user means specifically, search first. A general-knowledge answer that sounds right is the failure this library exists to prevent.
|
|
6912
|
+
|
|
6913
|
+
Do not search when the answer is already in front of you or out of scope: general knowledge (definitions, mathematics, established theory, public facts); anything answerable from this conversation, from an attached document, or from content already retrieved for you this turn; or a request to transform, summarize or reformat text the user has just supplied. If the library has already been searched on this turn, do not search it again for the same question - a repeat spends a round trip to return the same passages.
|
|
6914
|
+
|
|
6915
|
+
When a search does not turn up what was asked for, say so plainly rather than filling the gap from training data, and never imply an answer came from the user's documents when it did not.`;
|
|
6916
|
+
/**
|
|
6085
6917
|
* Default text for the formatting system message. Runtime fallback used by
|
|
6086
6918
|
* `includeHardcodedSystemMessage` (b4m-core/utils/src/llm/utils.ts) when the `FormatPromptTemplate`
|
|
6087
6919
|
* admin setting is blank; that setting's own default is intentionally '' - keep this the sole home.
|
|
@@ -6101,6 +6933,7 @@ z$1.enum([
|
|
|
6101
6933
|
"geminiDemoKey",
|
|
6102
6934
|
"xaiApiKey",
|
|
6103
6935
|
"moonshotApiKey",
|
|
6936
|
+
"deepseekApiKey",
|
|
6104
6937
|
"voyageApiKey",
|
|
6105
6938
|
"FirecrawlApiKey",
|
|
6106
6939
|
"FirecrawlApiUrl",
|
|
@@ -6114,6 +6947,8 @@ z$1.enum([
|
|
|
6114
6947
|
"ArtifactEmissionPrompt",
|
|
6115
6948
|
"HelpCenterPrompt",
|
|
6116
6949
|
"AbstentionPrompt",
|
|
6950
|
+
"WebSearchFreshnessPrompt",
|
|
6951
|
+
"KnowledgeBaseRetrievalPrompt",
|
|
6117
6952
|
"UseFormatPrompt",
|
|
6118
6953
|
"EnableQuestMaster",
|
|
6119
6954
|
"EnableQuestMasterDefault",
|
|
@@ -6132,11 +6967,11 @@ z$1.enum([
|
|
|
6132
6967
|
"EnableLattice",
|
|
6133
6968
|
"EnableLatticeDefault",
|
|
6134
6969
|
"EnableDataLakes",
|
|
6135
|
-
"EnableDataLakesDefault",
|
|
6136
6970
|
"EnableDataLakeSlackAdd",
|
|
6137
6971
|
"EnableDataLakeGroundingMode",
|
|
6138
6972
|
"EnableLakeMemory",
|
|
6139
6973
|
"EnableDataLakeVectorSearch",
|
|
6974
|
+
"EnableRetrievalSupersessionCollapse",
|
|
6140
6975
|
"PauseLakeConvergence",
|
|
6141
6976
|
"LakeConvergenceBulkChangeSharePct",
|
|
6142
6977
|
"EnforceLakeReadGrants",
|
|
@@ -6233,16 +7068,21 @@ z$1.enum([
|
|
|
6233
7068
|
"defaultEmbeddingModel",
|
|
6234
7069
|
"dataLakeSearchMaxFiles",
|
|
6235
7070
|
"dataLakeSearchMaxChunks",
|
|
7071
|
+
"dataLakeSearchMaxChunksPerFile",
|
|
6236
7072
|
"forcedRetrievalCharBudget",
|
|
7073
|
+
"lakeMemoryRecallK",
|
|
6237
7074
|
"kbSearchDefaultResults",
|
|
6238
7075
|
"kbSearchResultTokenBudget",
|
|
6239
7076
|
"kbSearchMinRelevancePct",
|
|
7077
|
+
"forcedRetrievalRelativeFloorPct",
|
|
7078
|
+
"forcedRetrievalMinSimilarityPct",
|
|
6240
7079
|
"dataLakeEmbeddingSpendEnabled",
|
|
6241
7080
|
"dataLakeEmbeddingBudgetPerRunUsd",
|
|
6242
7081
|
"dataLakeEmbeddingBudgetPerLakeUsd",
|
|
6243
7082
|
"dataLakeEmbeddingBudgetPerPeriodUsd",
|
|
6244
7083
|
"dataLakeEmbeddingBudgetPeriodHours",
|
|
6245
7084
|
"dataLakeEmbeddingMaxCallsPerMinute",
|
|
7085
|
+
"dataLakeEmbeddingMaxTokensPerMinute",
|
|
6246
7086
|
"dataLakeVectorizeChunkBatchSize",
|
|
6247
7087
|
"dataLakeEmbeddingTierMultiplierIndividual",
|
|
6248
7088
|
"dataLakeEmbeddingTierMultiplierOrganization",
|
|
@@ -6291,7 +7131,6 @@ z$1.enum([
|
|
|
6291
7131
|
"EnableBmPiDefault",
|
|
6292
7132
|
"EnableBmPiJira",
|
|
6293
7133
|
"EnableOptiHashi",
|
|
6294
|
-
"EnableOptiHashiDefault",
|
|
6295
7134
|
"EnableComputeSubmission",
|
|
6296
7135
|
"EnableFamilyCompute",
|
|
6297
7136
|
"EnableHybridCompute",
|
|
@@ -6319,6 +7158,7 @@ z$1.enum([
|
|
|
6319
7158
|
"modelDiscoveryAllowEgress",
|
|
6320
7159
|
"modelDiscoveryPriceBandPct",
|
|
6321
7160
|
"modelDiscoveryAutoRemap",
|
|
7161
|
+
"modelDiscoveryProbeNewModels",
|
|
6322
7162
|
"prReportRepo",
|
|
6323
7163
|
"prReportIdentityMap",
|
|
6324
7164
|
"prReportWebhookUrl",
|
|
@@ -6361,10 +7201,12 @@ const OrchestrationDefaultsSchema = z$1.object({
|
|
|
6361
7201
|
"mermaid_chart"
|
|
6362
7202
|
]),
|
|
6363
7203
|
/**
|
|
6364
|
-
* Tool names explicitly forbidden. Enforced as a final subtraction in
|
|
6365
|
-
* `pickEffectiveEnabledTools`
|
|
6366
|
-
*
|
|
6367
|
-
*
|
|
7204
|
+
* Tool names explicitly forbidden. Enforced in two places: as a final subtraction in
|
|
7205
|
+
* `pickEffectiveEnabledTools` (wins even over payload-pinned tools), and - for the two
|
|
7206
|
+
* delegation tools, which are injected as objects and never registered by name - at the
|
|
7207
|
+
* dependency gate in agentExecutor (`delegationOffer` withholds `agentStore` /
|
|
7208
|
+
* `dagDispatcher`). The name subtraction alone cannot reach those two; see
|
|
7209
|
+
* agentExecutor.sessionToolPolicy.
|
|
6368
7210
|
*
|
|
6369
7211
|
* Seeded with every tool that mutates user data (the spec's
|
|
6370
7212
|
* "anything tagged `mutates_user_data`"): destructive/overwriting filesystem
|
|
@@ -6442,8 +7284,30 @@ const DATA_LAKE_SEARCH_MAX_CHUNKS_DEFAULT = 1e5;
|
|
|
6442
7284
|
const DATA_LAKE_EMBEDDING_BUDGET_PER_LAKE_USD_MAX = 1e4;
|
|
6443
7285
|
const DATA_LAKE_EMBEDDING_BUDGET_PER_PERIOD_USD_MAX = 5e3;
|
|
6444
7286
|
const DATA_LAKE_EMBEDDING_MAX_CALLS_PER_MINUTE_MAX = 1e4;
|
|
7287
|
+
/**
|
|
7288
|
+
* The TOKEN half of the throughput cap, and the one that maps to what providers actually meter.
|
|
7289
|
+
* A call cap alone does not bound tokens: one call carries up to
|
|
7290
|
+
* DATA_LAKE_VECTORIZE_CHUNK_BATCH_SIZE_DEFAULT passages of DEFAULT_PASSAGE_TOKEN_TARGET tokens, so
|
|
7291
|
+
* 120 calls/min permits ~3.1M tokens/min - several times the smallest paid embeddings tier. The two
|
|
7292
|
+
* levers are complementary: calls/min bounds RPM, this bounds TPM, and a call must fit both.
|
|
7293
|
+
*
|
|
7294
|
+
* The default is deliberately LOW - it has to be safe on the smallest tier any deployment might
|
|
7295
|
+
* be on, including self-hosts nobody here can see. It is not a claim about what any particular
|
|
7296
|
+
* account can do, and reading it as one is the mistake to avoid: a provider tier is a property of
|
|
7297
|
+
* the provider organization, so it cannot be derived from this codebase at all.
|
|
7298
|
+
*
|
|
7299
|
+
* The real number is measurable per deployment: Admin -> Settings -> AI -> Data Lake Cost
|
|
7300
|
+
* Governance reads the configured provider's live ceiling (GET /api/admin/embedding-limits) and
|
|
7301
|
+
* shows it beside this lever, so an operator sets this from their own measured quota rather than
|
|
7302
|
+
* from a guess baked in here. Leave headroom below the measured ceiling for QUERY-side embedding,
|
|
7303
|
+
* which is exempt from this gate (see enforceEmbeddingSpendGate) and shares the same per-model
|
|
7304
|
+
* pool - a retrieval query must not queue behind a backfill.
|
|
7305
|
+
*/
|
|
7306
|
+
const DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_DEFAULT = 6e5;
|
|
7307
|
+
const DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_MAX = 5e7;
|
|
6445
7308
|
function makeNumberSetting(config) {
|
|
6446
7309
|
let numberSchema = z$1.coerce.number();
|
|
7310
|
+
if (config.int) numberSchema = numberSchema.int();
|
|
6447
7311
|
if (config.min !== void 0) numberSchema = numberSchema.min(config.min);
|
|
6448
7312
|
if (config.max !== void 0) numberSchema = numberSchema.max(config.max);
|
|
6449
7313
|
return {
|
|
@@ -7010,6 +7874,14 @@ const API_SERVICE_GROUPS = {
|
|
|
7010
7874
|
{
|
|
7011
7875
|
key: "AbstentionPrompt",
|
|
7012
7876
|
order: 11
|
|
7877
|
+
},
|
|
7878
|
+
{
|
|
7879
|
+
key: "WebSearchFreshnessPrompt",
|
|
7880
|
+
order: 12
|
|
7881
|
+
},
|
|
7882
|
+
{
|
|
7883
|
+
key: "KnowledgeBaseRetrievalPrompt",
|
|
7884
|
+
order: 13
|
|
7013
7885
|
}
|
|
7014
7886
|
]
|
|
7015
7887
|
},
|
|
@@ -7046,6 +7918,22 @@ const API_SERVICE_GROUPS = {
|
|
|
7046
7918
|
{
|
|
7047
7919
|
key: "kbSearchMinRelevancePct",
|
|
7048
7920
|
order: 7
|
|
7921
|
+
},
|
|
7922
|
+
{
|
|
7923
|
+
key: "lakeMemoryRecallK",
|
|
7924
|
+
order: 8
|
|
7925
|
+
},
|
|
7926
|
+
{
|
|
7927
|
+
key: "forcedRetrievalRelativeFloorPct",
|
|
7928
|
+
order: 9
|
|
7929
|
+
},
|
|
7930
|
+
{
|
|
7931
|
+
key: "forcedRetrievalMinSimilarityPct",
|
|
7932
|
+
order: 10
|
|
7933
|
+
},
|
|
7934
|
+
{
|
|
7935
|
+
key: "dataLakeSearchMaxChunksPerFile",
|
|
7936
|
+
order: 11
|
|
7049
7937
|
}
|
|
7050
7938
|
]
|
|
7051
7939
|
},
|
|
@@ -7080,16 +7968,20 @@ const API_SERVICE_GROUPS = {
|
|
|
7080
7968
|
order: 6
|
|
7081
7969
|
},
|
|
7082
7970
|
{
|
|
7083
|
-
key: "
|
|
7971
|
+
key: "dataLakeEmbeddingMaxTokensPerMinute",
|
|
7084
7972
|
order: 7
|
|
7085
7973
|
},
|
|
7086
7974
|
{
|
|
7087
|
-
key: "
|
|
7975
|
+
key: "dataLakeVectorizeChunkBatchSize",
|
|
7088
7976
|
order: 8
|
|
7089
7977
|
},
|
|
7090
7978
|
{
|
|
7091
|
-
key: "
|
|
7979
|
+
key: "dataLakeEmbeddingTierMultiplierIndividual",
|
|
7092
7980
|
order: 9
|
|
7981
|
+
},
|
|
7982
|
+
{
|
|
7983
|
+
key: "dataLakeEmbeddingTierMultiplierOrganization",
|
|
7984
|
+
order: 10
|
|
7093
7985
|
}
|
|
7094
7986
|
]
|
|
7095
7987
|
},
|
|
@@ -7149,6 +8041,16 @@ const API_SERVICE_GROUPS = {
|
|
|
7149
8041
|
order: 1
|
|
7150
8042
|
}]
|
|
7151
8043
|
},
|
|
8044
|
+
DEEPSEEK: {
|
|
8045
|
+
id: "deepseekAPIService",
|
|
8046
|
+
name: "DeepSeek Service",
|
|
8047
|
+
description: "DeepSeek API integration settings",
|
|
8048
|
+
icon: "AutoAwesome",
|
|
8049
|
+
settings: [{
|
|
8050
|
+
key: "deepseekApiKey",
|
|
8051
|
+
order: 1
|
|
8052
|
+
}]
|
|
8053
|
+
},
|
|
7152
8054
|
ANTHROPIC: {
|
|
7153
8055
|
id: "anthropicAPIService",
|
|
7154
8056
|
name: "Anthropic Service",
|
|
@@ -7486,10 +8388,6 @@ const API_SERVICE_GROUPS = {
|
|
|
7486
8388
|
key: "EnableOptiHashi",
|
|
7487
8389
|
order: 80
|
|
7488
8390
|
},
|
|
7489
|
-
{
|
|
7490
|
-
key: "EnableOptiHashiDefault",
|
|
7491
|
-
order: 81
|
|
7492
|
-
},
|
|
7493
8391
|
{
|
|
7494
8392
|
key: "EnableComputeSubmission",
|
|
7495
8393
|
order: 82
|
|
@@ -7852,6 +8750,10 @@ const API_SERVICE_GROUPS = {
|
|
|
7852
8750
|
{
|
|
7853
8751
|
key: "modelDiscoveryAutoRemap",
|
|
7854
8752
|
order: 6
|
|
8753
|
+
},
|
|
8754
|
+
{
|
|
8755
|
+
key: "modelDiscoveryProbeNewModels",
|
|
8756
|
+
order: 7
|
|
7855
8757
|
}
|
|
7856
8758
|
]
|
|
7857
8759
|
},
|
|
@@ -7937,6 +8839,16 @@ const settingsMap = {
|
|
|
7937
8839
|
group: API_SERVICE_GROUPS.MOONSHOT.id,
|
|
7938
8840
|
order: 1
|
|
7939
8841
|
}),
|
|
8842
|
+
deepseekApiKey: makeStringSetting({
|
|
8843
|
+
key: "deepseekApiKey",
|
|
8844
|
+
name: "DeepSeek API Key",
|
|
8845
|
+
defaultValue: "",
|
|
8846
|
+
description: "The global API Key for DeepSeek.",
|
|
8847
|
+
isSensitive: true,
|
|
8848
|
+
category: "AI",
|
|
8849
|
+
group: API_SERVICE_GROUPS.DEEPSEEK.id,
|
|
8850
|
+
order: 1
|
|
8851
|
+
}),
|
|
7940
8852
|
voyageApiKey: makeStringSetting({
|
|
7941
8853
|
key: "voyageApiKey",
|
|
7942
8854
|
name: "Voyage API Key",
|
|
@@ -7985,16 +8897,6 @@ const settingsMap = {
|
|
|
7985
8897
|
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
7986
8898
|
order: 88
|
|
7987
8899
|
}),
|
|
7988
|
-
EnableDataLakesDefault: makeBooleanSetting({
|
|
7989
|
-
key: "EnableDataLakesDefault",
|
|
7990
|
-
name: "Data Lakes: On by default for users",
|
|
7991
|
-
defaultValue: false,
|
|
7992
|
-
description: "When enabled, Data Lakes is active for users who have never explicitly toggled it.",
|
|
7993
|
-
category: "Experimental",
|
|
7994
|
-
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
7995
|
-
order: 89,
|
|
7996
|
-
dependsOn: "EnableDataLakes"
|
|
7997
|
-
}),
|
|
7998
8900
|
EnableDataLakeSlackAdd: makeBooleanSetting({
|
|
7999
8901
|
key: "EnableDataLakeSlackAdd",
|
|
8000
8902
|
name: "Data Lakes: Slack \"@datalake add\" path",
|
|
@@ -8019,7 +8921,7 @@ const settingsMap = {
|
|
|
8019
8921
|
key: "EnableLakeMemory",
|
|
8020
8922
|
name: "Data Lakes: Lake memory profile (extraction)",
|
|
8021
8923
|
defaultValue: false,
|
|
8022
|
-
description: "
|
|
8924
|
+
description: "Master gate for lake memory, on all three sides: LLM extraction of a data lake's documents into a durable memory profile, recall of that profile into chats grounded in the lake, and whether the per-lake opt-in is offered at all. Off by default (measurement rollout). Turning it off stops recall immediately and stops new extractions from being queued or picked up, though a run already in flight finishes its slice (bounded by the handler timeout). It is NOT destructive - each lake keeps its own opt-in and its built profile, so flipping this back on resumes where it left off. Erasing a profile is a separate, explicit per-lake action.",
|
|
8023
8925
|
category: "Experimental",
|
|
8024
8926
|
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8025
8927
|
order: 91,
|
|
@@ -8035,11 +8937,21 @@ const settingsMap = {
|
|
|
8035
8937
|
order: 92,
|
|
8036
8938
|
dependsOn: "EnableDataLakes"
|
|
8037
8939
|
}),
|
|
8940
|
+
EnableRetrievalSupersessionCollapse: makeBooleanSetting({
|
|
8941
|
+
key: "EnableRetrievalSupersessionCollapse",
|
|
8942
|
+
name: "Data Lakes: Collapse superseded members before ranking",
|
|
8943
|
+
defaultValue: false,
|
|
8944
|
+
description: "When a lake holds two generations of the same document (a re-upload, a Drive sync, a migration), rank only the newest and report the suppression. Off by default: the weakest identity tier is a bare file name, so two genuinely different documents sharing a name in one lake would collapse to one - turn this on only after checking the reported collapse counts on real lakes. Suppression is recoverable either way; a collapsed member is still reachable by id or name through retrieve_knowledge_content.",
|
|
8945
|
+
category: "Experimental",
|
|
8946
|
+
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8947
|
+
order: 97,
|
|
8948
|
+
dependsOn: "EnableDataLakes"
|
|
8949
|
+
}),
|
|
8038
8950
|
PauseLakeConvergence: makeBooleanSetting({
|
|
8039
8951
|
key: "PauseLakeConvergence",
|
|
8040
8952
|
name: "Data Lakes: Pause background convergence work",
|
|
8041
8953
|
defaultValue: false,
|
|
8042
|
-
description: "Kill switch for background data-lake ingestion work (convergence sweeps, rescue re-chunking) - NOT real-time user uploads, which are always honored. Off by default. Turn ON to halt in-flight background chunk/vectorize messages the next time the handler picks them up (a re-check inside the shared handler, so it takes effect on work already queued, not just the next scheduling pass). The platform value pauses every lake at once; a per-lake (or per-org / per-owner) override pauses a subset while the rest keep running. A platform-level flip applies immediately to lake-wide work and within ~5 min to per-lake-scoped work (settings cache).",
|
|
8954
|
+
description: "Kill switch for background data-lake ingestion work (convergence sweeps, rescue re-chunking) - NOT real-time user uploads, which are always honored. Off by default. Turn ON to halt in-flight background chunk/vectorize messages the next time the handler picks them up (a re-check inside the shared handler, so it takes effect on work already queued, not just the next scheduling pass). The platform value pauses every lake at once; a per-lake (or per-org / per-owner) override pauses a subset while the rest keep running - including overriding a platform-wide pause back OFF for one lake. Every producer honors the override, the global chunk rescue sweep included: it resolves each candidate against the lake it belongs to (#2157), so a file in no lake at all follows the platform value, which is correct for it. A platform-level flip applies immediately to lake-wide work and within ~5 min to per-lake-scoped work (settings cache).",
|
|
8043
8955
|
category: "Experimental",
|
|
8044
8956
|
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8045
8957
|
order: 93,
|
|
@@ -8069,8 +8981,8 @@ const settingsMap = {
|
|
|
8069
8981
|
EnforceLakeReadGrants: makeBooleanSetting({
|
|
8070
8982
|
key: "EnforceLakeReadGrants",
|
|
8071
8983
|
name: "Data Lakes: Enforce read-time grant resolution",
|
|
8072
|
-
defaultValue:
|
|
8073
|
-
description: "Read-time grant
|
|
8984
|
+
defaultValue: true,
|
|
8985
|
+
description: "Read-time grant resolution (#1673). ON is the shipped default: a persisted READER or ORG grant is resolved into the read decision, so a principal a lake was shared with can browse it, open it and ground on it. Resolution is purely ADDITIVE (legacy OR grant), so it takes no access away; this arm contains an ORG grant to the granting org, and expired rows never resolve. This is the standing KILL SWITCH for that arm, not a migration phase: turning it OFF returns to report-only, where the gate still resolves grants and logs where they WOULD change access ([lakeReadGrantCutover] lines) but the enforced decision falls back to the legacy owner/org/tag/entitlement/public rule - so those log lines are the diagnostic for a lake someone can no longer reach while the switch is off. Platform altitude on purpose: install-wide, not a per-lake lever. Tag and entitlement grants always resolve live and are never affected by this flag; only persisted reader/org rows are gated by it.",
|
|
8074
8986
|
category: "Experimental",
|
|
8075
8987
|
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8076
8988
|
order: 94,
|
|
@@ -8320,6 +9232,7 @@ const settingsMap = {
|
|
|
8320
9232
|
}),
|
|
8321
9233
|
DefaultChunkSize: makeNumberSetting({
|
|
8322
9234
|
key: "DefaultChunkSize",
|
|
9235
|
+
userReadable: true,
|
|
8323
9236
|
name: "Default Chunk Size",
|
|
8324
9237
|
defaultValue: 512,
|
|
8325
9238
|
min: 64,
|
|
@@ -8378,6 +9291,22 @@ const settingsMap = {
|
|
|
8378
9291
|
category: "AI",
|
|
8379
9292
|
order: 11
|
|
8380
9293
|
}),
|
|
9294
|
+
WebSearchFreshnessPrompt: makeStringSetting({
|
|
9295
|
+
key: "WebSearchFreshnessPrompt",
|
|
9296
|
+
name: "Web Search Freshness Prompt",
|
|
9297
|
+
defaultValue: WEB_SEARCH_FRESHNESS_PROMPT,
|
|
9298
|
+
description: "System prompt telling the model when to reach for web_search rather than answer from training data, and to state the as-of date of any time-sensitive fact. Injected only when the web_search tool is offered for the request - a model instructed to search without a search tool tends to claim it searched. Clearing this field turns the section OFF rather than restoring the built-in default, and it is the only off switch this section has; to get the stock wording back, paste it in. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else. After an upgrade, diff a saved copy against the built-in default: a saved copy pins the wording from whenever it was saved and will not pick up fixes made since.",
|
|
9299
|
+
category: "AI",
|
|
9300
|
+
order: 12
|
|
9301
|
+
}),
|
|
9302
|
+
KnowledgeBaseRetrievalPrompt: makeStringSetting({
|
|
9303
|
+
key: "KnowledgeBaseRetrievalPrompt",
|
|
9304
|
+
name: "Knowledge Base Retrieval Prompt",
|
|
9305
|
+
defaultValue: KNOWLEDGE_BASE_RETRIEVAL_PROMPT,
|
|
9306
|
+
description: "System prompt telling the model when to reach for search_knowledge_base rather than answer from training data, and when NOT to. Injected only when the search_knowledge_base tool is offered for the request - a model instructed to search a corpus it has no tool for tends to claim it searched. Clearing this field turns the section OFF rather than restoring the built-in default, and it is the only off switch this section has; to get the stock wording back, paste it in. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else. After an upgrade, diff a saved copy against the built-in default: a saved copy pins the wording from whenever it was saved and will not pick up fixes made since.",
|
|
9307
|
+
category: "AI",
|
|
9308
|
+
order: 13
|
|
9309
|
+
}),
|
|
8381
9310
|
UseFormatPrompt: makeBooleanSetting({
|
|
8382
9311
|
key: "UseFormatPrompt",
|
|
8383
9312
|
name: "Use Format Prompt",
|
|
@@ -8396,6 +9325,7 @@ const settingsMap = {
|
|
|
8396
9325
|
}),
|
|
8397
9326
|
pricePerCredit: makeNumberSetting({
|
|
8398
9327
|
key: "pricePerCredit",
|
|
9328
|
+
userReadable: true,
|
|
8399
9329
|
name: "Price Per Credit",
|
|
8400
9330
|
defaultValue: 50,
|
|
8401
9331
|
description: "The price per credit for purchasing credits.",
|
|
@@ -8473,7 +9403,8 @@ const settingsMap = {
|
|
|
8473
9403
|
name: "Referal Credits Amount",
|
|
8474
9404
|
defaultValue: 1e4,
|
|
8475
9405
|
description: "Credits to give to the referred user.",
|
|
8476
|
-
category: "Referrals"
|
|
9406
|
+
category: "Referrals",
|
|
9407
|
+
userReadable: true
|
|
8477
9408
|
}),
|
|
8478
9409
|
EnableReferralToEmail: makeBooleanSetting({
|
|
8479
9410
|
key: "EnableReferralToEmail",
|
|
@@ -8662,8 +9593,10 @@ const settingsMap = {
|
|
|
8662
9593
|
}),
|
|
8663
9594
|
MaxFileSize: makeNumberSetting({
|
|
8664
9595
|
key: "MaxFileSize",
|
|
9596
|
+
userReadable: true,
|
|
8665
9597
|
name: "Max File Size",
|
|
8666
9598
|
defaultValue: 30,
|
|
9599
|
+
min: 1,
|
|
8667
9600
|
description: "The maximum file size allowed for uploads in MB.",
|
|
8668
9601
|
category: "Knowledge",
|
|
8669
9602
|
group: API_SERVICE_GROUPS.KNOWLEDGE.id,
|
|
@@ -8770,6 +9703,7 @@ const settingsMap = {
|
|
|
8770
9703
|
}),
|
|
8771
9704
|
enforceCredits: makeBooleanSetting({
|
|
8772
9705
|
key: "enforceCredits",
|
|
9706
|
+
userReadable: true,
|
|
8773
9707
|
name: "Enforce Credits",
|
|
8774
9708
|
defaultValue: process.env.B4M_SELF_HOST === "true" ? false : true,
|
|
8775
9709
|
description: "Whether to enforce credits for users",
|
|
@@ -8786,6 +9720,7 @@ const settingsMap = {
|
|
|
8786
9720
|
}),
|
|
8787
9721
|
enableTeamPlan: makeBooleanSetting({
|
|
8788
9722
|
key: "enableTeamPlan",
|
|
9723
|
+
userReadable: true,
|
|
8789
9724
|
name: "Enable Team Plan",
|
|
8790
9725
|
defaultValue: false,
|
|
8791
9726
|
description: "Whether to enable team plans",
|
|
@@ -8891,7 +9826,8 @@ const settingsMap = {
|
|
|
8891
9826
|
description: "The global system prompt files to be used for AI model configuration.",
|
|
8892
9827
|
category: "AI",
|
|
8893
9828
|
group: API_SERVICE_GROUPS.OPENAI.id,
|
|
8894
|
-
order: 8
|
|
9829
|
+
order: 8,
|
|
9830
|
+
userReadable: true
|
|
8895
9831
|
}),
|
|
8896
9832
|
OpenWeatherKey: makeStringSetting({
|
|
8897
9833
|
key: "OpenWeatherKey",
|
|
@@ -9045,6 +9981,7 @@ const settingsMap = {
|
|
|
9045
9981
|
}),
|
|
9046
9982
|
MaxContentLength: makeNumberSetting({
|
|
9047
9983
|
key: "MaxContentLength",
|
|
9984
|
+
userReadable: true,
|
|
9048
9985
|
name: "Max Content Length",
|
|
9049
9986
|
defaultValue: 5e4,
|
|
9050
9987
|
description: "The maximum character length for file content displayed in workbench (truncated if larger).",
|
|
@@ -9210,9 +10147,10 @@ const settingsMap = {
|
|
|
9210
10147
|
}),
|
|
9211
10148
|
defaultEmbeddingModel: makeStringSetting({
|
|
9212
10149
|
key: "defaultEmbeddingModel",
|
|
10150
|
+
userReadable: true,
|
|
9213
10151
|
name: "Default Embedding Model",
|
|
9214
10152
|
defaultValue: defaultEmbeddingModelForEnv(),
|
|
9215
|
-
description: "The default embedding model to use",
|
|
10153
|
+
description: "The default embedding model to use. Changing it changes the SCALE of every similarity score in the system, so relevance floors do not carry across: a floor tuned for one model can sit above the entire range of another and reject everything. The server handles this for you on the floors it ships, applying the value measured for whichever model your documents are actually embedded with - but if you have set Forced Retrieval Absolute Floor by hand, re-measure it after changing this. Existing documents keep their old vectors and are only comparable to a query embedded the same way, so a change here needs a re-embed to take full effect; until then each set of documents is searched with the model it was indexed under.",
|
|
9216
10154
|
category: "AI",
|
|
9217
10155
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9218
10156
|
options: [
|
|
@@ -9231,11 +10169,7 @@ const settingsMap = {
|
|
|
9231
10169
|
category: "AI",
|
|
9232
10170
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9233
10171
|
order: 2,
|
|
9234
|
-
scope: { settableAt: [
|
|
9235
|
-
"organization",
|
|
9236
|
-
"owner",
|
|
9237
|
-
"lake"
|
|
9238
|
-
] }
|
|
10172
|
+
scope: { settableAt: ["organization", "owner"] }
|
|
9239
10173
|
}),
|
|
9240
10174
|
dataLakeSearchMaxChunks: makeNumberSetting({
|
|
9241
10175
|
key: "dataLakeSearchMaxChunks",
|
|
@@ -9246,11 +10180,18 @@ const settingsMap = {
|
|
|
9246
10180
|
category: "AI",
|
|
9247
10181
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9248
10182
|
order: 3,
|
|
9249
|
-
scope: { settableAt: [
|
|
9250
|
-
|
|
9251
|
-
|
|
9252
|
-
|
|
9253
|
-
|
|
10183
|
+
scope: { settableAt: ["organization", "owner"] }
|
|
10184
|
+
}),
|
|
10185
|
+
dataLakeSearchMaxChunksPerFile: makeNumberSetting({
|
|
10186
|
+
key: "dataLakeSearchMaxChunksPerFile",
|
|
10187
|
+
name: "Data Lake Search Max Chunks Per Document",
|
|
10188
|
+
defaultValue: 0,
|
|
10189
|
+
min: 0,
|
|
10190
|
+
description: "Most chunks from any ONE source document a data-lake semantic search may return in its top-K. A diversity guard for CONTESTED slots: where several documents answer the question, it stops the best-scoring one from taking slots the others could have filled. It is NOT a fix for severe crowding - the cap redistributes only among the candidates retrieval already returned, so a document that supplies enough of the top-scoring chunks to fill that pool on its own is one the cap cannot change at all. On a corpus of book-length documents, expect enabling this to change little beyond widening the vector-search request. 0 (default) disables the cap, byte-identical to behavior before this setting existed. The cap never SHRINKS a result set - once the spread-out picks are in, any slots still open are backfilled with the highest-scoring chunks the cap held back, so a lake whose only match is one document still returns a full top-K. A value at or above the result count is also a no-op, since nothing can ever be held back. Below it, each retrieval stream's candidate pool is widened to a fixed multiple of the result count so the cap has a spread to choose from. The scanned corpus itself does not grow (that is bounded separately), but the vector-search backends are asked for that many more matches, and a larger in-memory ranking pool costs some CPU. 2-3 is the useful range; 1 serves one passage per document, which suits a corpus of many short documents and starves a question whose answer spans one long one. The chat knowledge-base path ranks more passages than it serves, so it applies the cap a second time at the count it actually serves - otherwise the spread-out picks, which are by definition the lowest-scoring ones admitted, would land in the passages that path discards.",
|
|
10191
|
+
category: "AI",
|
|
10192
|
+
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
10193
|
+
order: 11,
|
|
10194
|
+
scope: { settableAt: ["organization", "owner"] }
|
|
9254
10195
|
}),
|
|
9255
10196
|
forcedRetrievalCharBudget: makeNumberSetting({
|
|
9256
10197
|
key: "forcedRetrievalCharBudget",
|
|
@@ -9258,10 +10199,11 @@ const settingsMap = {
|
|
|
9258
10199
|
defaultValue: FORCED_RETRIEVAL_CHAR_BUDGET_DEFAULT,
|
|
9259
10200
|
min: 1e3,
|
|
9260
10201
|
max: 1e5,
|
|
9261
|
-
description: "Total characters of retrieved chunk text injected into a Data-Lake-mode turn. Measured saturating on every turn against a 47-document lake, so this is the binding constraint on how much of a corpus reaches the model - not the relevance floor. Raising it admits more passages at the cost of prompt tokens and latency on every Data-Lake turn; it is NOT automatically better, since more context can dilute ranking.
|
|
10202
|
+
description: "Total characters of retrieved chunk text injected into a Data-Lake-mode turn. Measured saturating on every turn against a 47-document lake, so this is the binding constraint on how much of a corpus reaches the model - not the relevance floor. Raising it admits more passages at the cost of prompt tokens and latency on every Data-Lake turn; it is NOT automatically better, since more context can dilute ranking. Overridable per organization and per owner, the same altitude as the two relevance floors resolved alongside it on the same turn.",
|
|
9262
10203
|
category: "AI",
|
|
9263
10204
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9264
|
-
order: 4
|
|
10205
|
+
order: 4,
|
|
10206
|
+
scope: { settableAt: ["organization", "owner"] }
|
|
9265
10207
|
}),
|
|
9266
10208
|
kbSearchDefaultResults: makeNumberSetting({
|
|
9267
10209
|
key: "kbSearchDefaultResults",
|
|
@@ -9269,7 +10211,7 @@ const settingsMap = {
|
|
|
9269
10211
|
defaultValue: 5,
|
|
9270
10212
|
min: 1,
|
|
9271
10213
|
max: 10,
|
|
9272
|
-
description: "Passages the search_knowledge_base tool returns when a model call omits max_results, which is most calls. This is the exact bound while kbSearchResultTokenBudget is unset (0). Once a token budget is set, it takes over as the primary bound for search results (this setting's own value is then unused there, though it still governs the keyword-search fallback, and the count served if token pricing itself fails). Does NOT raise the tool's hard ceiling of 10 passages per call - a model that reads max_results up to 10 from its own tool schema won't ask for more than that regardless of this setting.",
|
|
10214
|
+
description: "Passages the search_knowledge_base tool returns when a model call omits max_results, which is most calls. This is the exact bound while kbSearchResultTokenBudget is unset (0). Once a token budget is set, it takes over as the primary bound for search results (this setting's own value is then unused there, though it still governs the keyword-search fallback, and the count served if token pricing itself fails). Does NOT raise the tool's hard ceiling of 10 passages per call - a model that reads max_results up to 10 from its own tool schema won't ask for more than that regardless of this setting. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else.",
|
|
9273
10215
|
category: "AI",
|
|
9274
10216
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9275
10217
|
order: 5,
|
|
@@ -9281,7 +10223,7 @@ const settingsMap = {
|
|
|
9281
10223
|
defaultValue: 0,
|
|
9282
10224
|
min: 0,
|
|
9283
10225
|
max: 2e4,
|
|
9284
|
-
description: "Approximate tokens of served passage TEXT (post-trim, post-clip - what the model actually receives, not the raw stored chunk) the search_knowledge_base tool may return in one call. Counted with a fixed tokenizer as a proxy, not billed against any specific model. Replaces a passage count as the primary bound once set, since it is invariant to chunk size - a lake chunked smaller no longer silently returns less material for the same setting. 0 (default) disables it: search_knowledge_base then serves exactly kbSearchDefaultResults passages, unchanged from before this setting existed. The FIRST matching passage is always returned even if it alone exceeds the budget - a search that found something never returns nothing.",
|
|
10226
|
+
description: "Approximate tokens of served passage TEXT (post-trim, post-clip - what the model actually receives, not the raw stored chunk) the search_knowledge_base tool may return in one call. Counted with a fixed tokenizer as a proxy, not billed against any specific model. Replaces a passage count as the primary bound once set, since it is invariant to chunk size - a lake chunked smaller no longer silently returns less material for the same setting. 0 (default) disables it: search_knowledge_base then serves exactly kbSearchDefaultResults passages, unchanged from before this setting existed. The FIRST matching passage is always returned even if it alone exceeds the budget - a search that found something never returns nothing. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else.",
|
|
9285
10227
|
category: "AI",
|
|
9286
10228
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9287
10229
|
order: 6,
|
|
@@ -9293,12 +10235,50 @@ const settingsMap = {
|
|
|
9293
10235
|
defaultValue: 0,
|
|
9294
10236
|
min: 0,
|
|
9295
10237
|
max: 100,
|
|
9296
|
-
description: "Minimum cosine relevance, as a percent, a passage must clear to be returned by search_knowledge_base. 0 (default) matches current behavior (no relevance floor beyond a non-negative cosine score). Raising it lets breadth adapt per query - a narrow question can return fewer, more relevant passages instead of always padding out to the configured count. Cosine similarity is not comparable across embedding models: a floor tuned for one model can filter out an entire alternate model, when a lake mixes embedding models, more aggressively than intended. Start low and raise gradually while watching the tool's own retrieval-skipped notices.",
|
|
10238
|
+
description: "Minimum cosine relevance, as a percent, a passage must clear to be returned by search_knowledge_base. 0 (default) matches current behavior (no relevance floor beyond a non-negative cosine score). Raising it lets breadth adapt per query - a narrow question can return fewer, more relevant passages instead of always padding out to the configured count. Cosine similarity is not comparable across embedding models: a floor tuned for one model can filter out an entire alternate model, when a lake mixes embedding models, more aggressively than intended. Start low and raise gradually while watching the tool's own retrieval-skipped notices. A change is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min (one cache TTL) everywhere else.",
|
|
9297
10239
|
category: "AI",
|
|
9298
10240
|
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
9299
10241
|
order: 7,
|
|
9300
10242
|
scope: { settableAt: ["organization", "owner"] }
|
|
9301
10243
|
}),
|
|
10244
|
+
lakeMemoryRecallK: makeNumberSetting({
|
|
10245
|
+
key: "lakeMemoryRecallK",
|
|
10246
|
+
name: "Lake Memory Belief Budget",
|
|
10247
|
+
defaultValue: 24,
|
|
10248
|
+
min: 1,
|
|
10249
|
+
max: 200,
|
|
10250
|
+
int: true,
|
|
10251
|
+
description: "Most beliefs the lake memory hot-card injects on a Data-Lake-mode turn, shared across every lake in scope. Recall still applies its cosine floor and the source-reachability gate first, so raising this does not admit low-quality beliefs - it raises the ceiling on how many QUALIFYING beliefs can actually be used, which was pinned at 8 (inherited from personal-memento recall) on no evidence beyond that inheritance. The sibling lever on the same turn is Forced Retrieval Char Budget, which governs raw chunk text rather than extracted beliefs. Platform-only for now, unlike that sibling: this read goes through plain getSettingsValue, which ignores settableAt, so a scope block here would be silently inert - every override written against it would resolve to nothing. Pointing the read at the scoped resolver is the prerequisite, not extra metadata.",
|
|
10252
|
+
category: "AI",
|
|
10253
|
+
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
10254
|
+
order: 8
|
|
10255
|
+
}),
|
|
10256
|
+
forcedRetrievalRelativeFloorPct: makeNumberSetting({
|
|
10257
|
+
key: "forcedRetrievalRelativeFloorPct",
|
|
10258
|
+
name: "Forced Retrieval Relative Floor (%)",
|
|
10259
|
+
defaultValue: 85,
|
|
10260
|
+
min: 0,
|
|
10261
|
+
max: 100,
|
|
10262
|
+
int: true,
|
|
10263
|
+
description: "How close to the best-scoring passage of the SAME turn a chunk must score to be injected on a Data-Lake-mode turn, as a percent of that top score. This is the floor that ranks; the absolute floor below only rejects. Unlike an absolute cosine line, it moves with the turn, so it keeps working when a corpus or an embedding model puts the whole score band somewhere else. Raising it injects fewer, more sharply-ranked passages and leaves char budget unspent; lowering it admits more of the tail. 0 disables the relative floor and leaves the absolute one as the only gate (the pre-#2497 behavior). The default is behavior-preserving rather than tuned: it admits everything the absolute floor admitted on the measured band, so it changes nothing until raised. Tune it AFTER an embedding-model change, never before - a migration shifts the band any value fitted to today would have been chosen against.",
|
|
10264
|
+
category: "AI",
|
|
10265
|
+
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
10266
|
+
order: 9,
|
|
10267
|
+
scope: { settableAt: ["organization", "owner"] }
|
|
10268
|
+
}),
|
|
10269
|
+
forcedRetrievalMinSimilarityPct: makeNumberSetting({
|
|
10270
|
+
key: "forcedRetrievalMinSimilarityPct",
|
|
10271
|
+
name: "Forced Retrieval Absolute Floor (%)",
|
|
10272
|
+
defaultValue: 75,
|
|
10273
|
+
min: 1,
|
|
10274
|
+
max: 100,
|
|
10275
|
+
int: true,
|
|
10276
|
+
description: `Absolute minimum cosine similarity, as a percent, a chunk must clear to be injected on a Data-Lake-mode turn. This is a sanity floor for genuinely unrelated content, NOT the ranking gate - the relative floor above does the ranking. LEAVE IT AT 75 UNLESS YOU HAVE MEASURED YOUR OWN CORPUS: a raw cosine means nothing outside the embedding model it was fitted to, so while this reads 75 the server ignores it and applies the floor measured for whichever model your documents are actually embedded with (${forcedRetrievalFloorsBySpaceSummary}, and no absolute floor at all for a model nobody has measured - the relative floor still applies). Set any other value and the server uses exactly that, in every space, which is yours to get right: 75 against text-embedding-3-small sits above that band entirely and returns nothing on every query. Where this floor lands inside your band decides a lot - on one measured corpus 74 / 75 / 76 swung recall 91% / 65% / 40% - and the same 75 that is a cliff on one lake rejects nothing at all on another. Re-measure after changing the embedding model; the sweep tool is packages/scripts/retrieval/forcedFloorSweep.ts.`,
|
|
10277
|
+
category: "AI",
|
|
10278
|
+
group: API_SERVICE_GROUPS.EMBEDDING.id,
|
|
10279
|
+
order: 10,
|
|
10280
|
+
scope: { settableAt: ["organization", "owner"] }
|
|
10281
|
+
}),
|
|
9302
10282
|
LakeAccessAuditRetentionDays: makeNumberSetting({
|
|
9303
10283
|
key: "LakeAccessAuditRetentionDays",
|
|
9304
10284
|
name: "Lake Access Audit Retention (days)",
|
|
@@ -9334,6 +10314,7 @@ const settingsMap = {
|
|
|
9334
10314
|
}),
|
|
9335
10315
|
dataLakeEmbeddingSpendEnabled: makeBooleanSetting({
|
|
9336
10316
|
key: "dataLakeEmbeddingSpendEnabled",
|
|
10317
|
+
userReadable: true,
|
|
9337
10318
|
name: "Data Lake Embedding Spend Enabled",
|
|
9338
10319
|
defaultValue: true,
|
|
9339
10320
|
description: "Master switch for data-lake embedding spend (ingestion, reprocessing, convergence). Off halts all provider embedding calls on those paths; cached embeddings still apply.",
|
|
@@ -9343,6 +10324,7 @@ const settingsMap = {
|
|
|
9343
10324
|
}),
|
|
9344
10325
|
dataLakeEmbeddingBudgetPerRunUsd: makeNumberSetting({
|
|
9345
10326
|
key: "dataLakeEmbeddingBudgetPerRunUsd",
|
|
10327
|
+
userReadable: true,
|
|
9346
10328
|
name: "Embedding Budget Per Run (USD)",
|
|
9347
10329
|
defaultValue: 5,
|
|
9348
10330
|
min: 0,
|
|
@@ -9396,6 +10378,17 @@ const settingsMap = {
|
|
|
9396
10378
|
group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
|
|
9397
10379
|
order: 6
|
|
9398
10380
|
}),
|
|
10381
|
+
dataLakeEmbeddingMaxTokensPerMinute: makeNumberSetting({
|
|
10382
|
+
key: "dataLakeEmbeddingMaxTokensPerMinute",
|
|
10383
|
+
name: "Embedding Max Tokens Per Minute",
|
|
10384
|
+
defaultValue: DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_DEFAULT,
|
|
10385
|
+
min: 0,
|
|
10386
|
+
max: DATA_LAKE_EMBEDDING_MAX_TOKENS_PER_MINUTE_MAX,
|
|
10387
|
+
description: "Most provider embedding TOKENS per minute across all data-lake work, which is the quantity providers actually meter. The calls-per-minute lever alone does not bound this: one call carries a whole batch of passages. Set it from your provider dashboard TPM, leaving headroom for query-side embedding (exempt, so a search never queues behind a backfill). 0 stops all calls.",
|
|
10388
|
+
category: "AI",
|
|
10389
|
+
group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
|
|
10390
|
+
order: 7
|
|
10391
|
+
}),
|
|
9399
10392
|
dataLakeVectorizeChunkBatchSize: makeNumberSetting({
|
|
9400
10393
|
key: "dataLakeVectorizeChunkBatchSize",
|
|
9401
10394
|
name: "Vectorize Chunk Batch Size",
|
|
@@ -9405,7 +10398,7 @@ const settingsMap = {
|
|
|
9405
10398
|
description: "How many chunks the chunk handler packs into one vectorize-queue message. Smaller batches smooth the fan-out; not a spend value, so min 1.",
|
|
9406
10399
|
category: "AI",
|
|
9407
10400
|
group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
|
|
9408
|
-
order:
|
|
10401
|
+
order: 8
|
|
9409
10402
|
}),
|
|
9410
10403
|
dataLakeEmbeddingTierMultiplierIndividual: makeNumberSetting({
|
|
9411
10404
|
key: "dataLakeEmbeddingTierMultiplierIndividual",
|
|
@@ -9416,7 +10409,7 @@ const settingsMap = {
|
|
|
9416
10409
|
description: "Scales the per-run and per-lake embedding budgets for lakes owned by an individual user. 1 means those lakes get exactly the configured budgets; 0 stops them spending at all. The effective budget is still capped by the same hard rail as the untiered value.",
|
|
9417
10410
|
category: "AI",
|
|
9418
10411
|
group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
|
|
9419
|
-
order:
|
|
10412
|
+
order: 9
|
|
9420
10413
|
}),
|
|
9421
10414
|
dataLakeEmbeddingTierMultiplierOrganization: makeNumberSetting({
|
|
9422
10415
|
key: "dataLakeEmbeddingTierMultiplierOrganization",
|
|
@@ -9427,7 +10420,7 @@ const settingsMap = {
|
|
|
9427
10420
|
description: "Scales the per-run and per-lake embedding budgets for lakes owned by an organization, which serve a whole team rather than one person. 0 stops org-owned lakes spending at all. The effective budget is still capped by the same hard rail as the untiered value.",
|
|
9428
10421
|
category: "AI",
|
|
9429
10422
|
group: API_SERVICE_GROUPS.DATA_LAKE_COST.id,
|
|
9430
|
-
order:
|
|
10423
|
+
order: 10
|
|
9431
10424
|
}),
|
|
9432
10425
|
slackSigningSecret: makeStringSetting({
|
|
9433
10426
|
key: "slackSigningSecret",
|
|
@@ -9529,6 +10522,7 @@ const settingsMap = {
|
|
|
9529
10522
|
}),
|
|
9530
10523
|
enableVoiceSession: makeBooleanSetting({
|
|
9531
10524
|
key: "enableVoiceSession",
|
|
10525
|
+
userReadable: true,
|
|
9532
10526
|
name: "Enable Voice Session",
|
|
9533
10527
|
defaultValue: false,
|
|
9534
10528
|
description: "Whether to enable the voice session.",
|
|
@@ -9538,6 +10532,7 @@ const settingsMap = {
|
|
|
9538
10532
|
}),
|
|
9539
10533
|
voiceV2Enabled: makeBooleanSetting({
|
|
9540
10534
|
key: "voiceV2Enabled",
|
|
10535
|
+
userReadable: true,
|
|
9541
10536
|
name: "Enable Voice v2 (Model-Agnostic)",
|
|
9542
10537
|
defaultValue: false,
|
|
9543
10538
|
description: "Gate for the Voice v2 feature (ElevenLabs Conversational AI + any B4M reasoning model). When disabled, /api/voice/v2/sessions returns 403.",
|
|
@@ -9557,6 +10552,7 @@ const settingsMap = {
|
|
|
9557
10552
|
}),
|
|
9558
10553
|
voiceSessionAiVoice: makeStringSetting({
|
|
9559
10554
|
key: "voiceSessionAiVoice",
|
|
10555
|
+
userReadable: true,
|
|
9560
10556
|
name: "Default Assistant Voice",
|
|
9561
10557
|
defaultValue: "alloy",
|
|
9562
10558
|
description: "The default voice for the assistant in the voice session.",
|
|
@@ -9858,16 +10854,6 @@ const settingsMap = {
|
|
|
9858
10854
|
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
9859
10855
|
order: 80
|
|
9860
10856
|
}),
|
|
9861
|
-
EnableOptiHashiDefault: makeBooleanSetting({
|
|
9862
|
-
key: "EnableOptiHashiDefault",
|
|
9863
|
-
name: "OptiHashi: On by default for users",
|
|
9864
|
-
defaultValue: false,
|
|
9865
|
-
description: "When enabled, OptiHashi is active for users who have never explicitly toggled it.",
|
|
9866
|
-
category: "Experimental",
|
|
9867
|
-
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
9868
|
-
order: 81,
|
|
9869
|
-
dependsOn: "EnableOptiHashi"
|
|
9870
|
-
}),
|
|
9871
10857
|
EnableLibreOncology: makeBooleanSetting({
|
|
9872
10858
|
key: "EnableLibreOncology",
|
|
9873
10859
|
name: "Enable LibreOncology",
|
|
@@ -10068,6 +11054,7 @@ const settingsMap = {
|
|
|
10068
11054
|
}),
|
|
10069
11055
|
orchestrationDefaults: makeObjectSetting({
|
|
10070
11056
|
key: "orchestrationDefaults",
|
|
11057
|
+
userReadable: true,
|
|
10071
11058
|
name: "Agent Orchestration Defaults",
|
|
10072
11059
|
defaultValue: OrchestrationDefaultsSchema.parse({}),
|
|
10073
11060
|
description: "Default ReAct profile for agentless executions (#8922). Drives allowed/denied tools, iteration ceilings, default thoroughness, and fallback models when the agent_executor is invoked without a persisted IAgent (e.g. the upcoming Agent-mode toggle).",
|
|
@@ -10138,6 +11125,15 @@ const settingsMap = {
|
|
|
10138
11125
|
group: API_SERVICE_GROUPS.MODEL_DISCOVERY.id,
|
|
10139
11126
|
order: 6
|
|
10140
11127
|
}),
|
|
11128
|
+
modelDiscoveryProbeNewModels: makeBooleanSetting({
|
|
11129
|
+
key: "modelDiscoveryProbeNewModels",
|
|
11130
|
+
name: "Probe New Models for Dispatch",
|
|
11131
|
+
defaultValue: true,
|
|
11132
|
+
description: "Lets a write-mode run spend a forced one-tool call on a newly discovered OpenAI model to verify which token parameter and tool transport it takes, and turn its tools on. Off leaves every new OpenAI model with tools withheld until an operator writes the dispatch profile by hand.",
|
|
11133
|
+
category: "AI",
|
|
11134
|
+
group: API_SERVICE_GROUPS.MODEL_DISCOVERY.id,
|
|
11135
|
+
order: 7
|
|
11136
|
+
}),
|
|
10141
11137
|
prReportRepo: makeStringSetting({
|
|
10142
11138
|
key: "prReportRepo",
|
|
10143
11139
|
name: "PR Report Repository",
|
|
@@ -10230,7 +11226,8 @@ const ModelTelemetrySchema = z$1.object({
|
|
|
10230
11226
|
"google",
|
|
10231
11227
|
"xai",
|
|
10232
11228
|
"ollama",
|
|
10233
|
-
"moonshot"
|
|
11229
|
+
"moonshot",
|
|
11230
|
+
"deepseek"
|
|
10234
11231
|
]),
|
|
10235
11232
|
fallbackUsed: z$1.boolean(),
|
|
10236
11233
|
fallbackReason: z$1.string().optional(),
|
|
@@ -10247,7 +11244,8 @@ const SystemPromptDetailSchema = z$1.object({
|
|
|
10247
11244
|
"user",
|
|
10248
11245
|
"project",
|
|
10249
11246
|
"session",
|
|
10250
|
-
"org"
|
|
11247
|
+
"org",
|
|
11248
|
+
"caller"
|
|
10251
11249
|
]),
|
|
10252
11250
|
/** e.g., "date_context", "tool_guidance" */
|
|
10253
11251
|
name: z$1.string(),
|
|
@@ -10641,6 +11639,7 @@ const PromptMetaContextSchema = z$1.object({
|
|
|
10641
11639
|
}).optional(),
|
|
10642
11640
|
lakeMemory: z$1.object({
|
|
10643
11641
|
beliefCount: z$1.number(),
|
|
11642
|
+
beliefBudget: z$1.number().optional(),
|
|
10644
11643
|
dataLakeTags: z$1.array(z$1.string())
|
|
10645
11644
|
}).optional(),
|
|
10646
11645
|
contextWindowUsage: z$1.object({
|
|
@@ -10685,7 +11684,15 @@ const PromptMetaPerformanceSchema = z$1.object({
|
|
|
10685
11684
|
totalResponseTime: z$1.number().optional(),
|
|
10686
11685
|
contextRetrievalTime: z$1.number().optional(),
|
|
10687
11686
|
modelInferenceTime: z$1.number().optional(),
|
|
11687
|
+
/**
|
|
11688
|
+
* Time to First Visible Token: elapsed ms until the first chunk the user can actually
|
|
11689
|
+
* see. Left unset when a turn streamed nothing visible (thinking-only, or a turn that
|
|
11690
|
+
* errored before answering), so absence reads as "never rendered" rather than as fast.
|
|
11691
|
+
* Pair with firstChunkTime to tell a slow model from a long hidden-reasoning window.
|
|
11692
|
+
*/
|
|
10688
11693
|
firstTokenTime: z$1.number().optional(),
|
|
11694
|
+
/** Elapsed ms until the first chunk of any kind, including a hidden thinking block. */
|
|
11695
|
+
firstChunkTime: z$1.number().optional(),
|
|
10689
11696
|
clientFirstTokenTime: z$1.number().optional(),
|
|
10690
11697
|
streamingPerformance: z$1.object({
|
|
10691
11698
|
chunkCount: z$1.number().optional(),
|
|
@@ -10766,11 +11773,16 @@ const CitableSourceSchema = z$1.object({
|
|
|
10766
11773
|
* zero-result retrieval, so a turn that legitimately found nothing is indistinguishable from one
|
|
10767
11774
|
* where retrieval never ran at all.
|
|
10768
11775
|
*
|
|
10769
|
-
*
|
|
10770
|
-
* more precise: `citables.filter(c => c.type === 'document')` is deduped by id/url/title in
|
|
11776
|
+
* Holds NO chunk/document identifiers, and no DOCUMENT count. A document count already exists and
|
|
11777
|
+
* is more precise: `citables.filter(c => c.type === 'document')` is deduped by id/url/title in
|
|
10771
11778
|
* `applyQuestStatusChanges`, while this shape cannot dedupe (no identifiers to dedupe by) and
|
|
10772
|
-
* would have to sum - producing a second, disagreeing number for the same question.
|
|
10773
|
-
*
|
|
11779
|
+
* would have to sum - producing a second, disagreeing number for the same question.
|
|
11780
|
+
*
|
|
11781
|
+
* `injected` below is NOT that number and does not reopen it: it counts PASSAGES and characters,
|
|
11782
|
+
* neither of which `citables` can express - one document contributes many passages, and a turn
|
|
11783
|
+
* that injected nothing emits no citable to count at all. Similarity scores otherwise live on
|
|
11784
|
+
* `LakeAccessEvent`; the single `injected.topScore` is here because that row is written only on a
|
|
11785
|
+
* turn that grounded, so it cannot carry the near-miss score of a turn that grounded on nothing.
|
|
10774
11786
|
*
|
|
10775
11787
|
* CAUTION, not a guarantee: the absence of chunk/document identifiers is what keeps this shape
|
|
10776
11788
|
* OUT of `promptMetaRedaction.ts`'s scope (that helper is a functionCalls-only denylist and would
|
|
@@ -10789,28 +11801,351 @@ const CitableSourceSchema = z$1.object({
|
|
|
10789
11801
|
*
|
|
10790
11802
|
* Absent-or-fully-present, matching `lakeMemory` above - see the Mongoose-side subSchema comment
|
|
10791
11803
|
* in QuestModel.ts for why partial-write and default-array shapes are unsafe here.
|
|
11804
|
+
*
|
|
11805
|
+
* WIDER THAN ITS NAME SUGGESTS as of `mode` (#1394). The field is no longer written only when
|
|
11806
|
+
* retrieval ran: it is now seeded on every turn that could have retrieved (forced retrieval
|
|
11807
|
+
* enabled, or the knowledge tool offered), so `attempted: false` is a recorded fact rather than
|
|
11808
|
+
* an absence to be inferred. Presence therefore means "this turn was in a position to retrieve",
|
|
11809
|
+
* and turns with no knowledge in scope still carry no field at all. The distinction matters to a
|
|
11810
|
+
* rollup: absence is now ambiguous between "not a retrieval turn" and "written before this
|
|
11811
|
+
* existed", which is why `mode` documents its own date-bounding requirement.
|
|
10792
11812
|
*/
|
|
10793
11813
|
const RetrievalSummarySchema = z$1.object({
|
|
10794
11814
|
/** True once a retrieval-capable surface actually ran (not merely offered) this turn. */
|
|
10795
11815
|
attempted: z$1.boolean(),
|
|
10796
11816
|
/**
|
|
11817
|
+
* Present if and only if `attempted` is true - an outcome describes a run, and a turn that
|
|
11818
|
+
* never ran retrieval has none. A reader testing for a specific value is unaffected (absence
|
|
11819
|
+
* is not any of them); a reader switching exhaustively must handle undefined.
|
|
11820
|
+
*
|
|
10797
11821
|
* 'ok' - ran, whether or not anything came back (the zero case is a legitimate 'ok').
|
|
10798
11822
|
* 'no_lakes' - ran but the user had no entitled/selected lake in scope.
|
|
10799
|
-
* '
|
|
10800
|
-
*
|
|
10801
|
-
*
|
|
10802
|
-
*
|
|
10803
|
-
*
|
|
11823
|
+
* 'not_indexed' - ran to completion having compared nothing: the corpus in scope carries no
|
|
11824
|
+
* usable vector (never indexed, or embedded with a foreign model), so no passage was ever
|
|
11825
|
+
* scored against the query. Distinct from 'ok' because the library was not searched at all,
|
|
11826
|
+
* and reporting that as a topical zero ("your documents do not cover this") is exactly the
|
|
11827
|
+
* confident-wrong-answer this field exists to catch. Distinct from 'failed' because nothing
|
|
11828
|
+
* broke: the remedy is re-vectorizing, which the corpus owner can do themselves, and a retry
|
|
11829
|
+
* never helps.
|
|
11830
|
+
* COVERAGE: recorded by forced retrieval (KnowledgeRetrievalFeature's `scoredCount === 0`
|
|
11831
|
+
* exit) and by knowledgeBaseSearch, whose semantic arms carry the same verdict through to the
|
|
11832
|
+
* keyword arm's write - the two agree on "not one passage was compared against the query",
|
|
11833
|
+
* not on any withholding flag, so a relevance floor that emptied a real search and a partial
|
|
11834
|
+
* withholding alongside a real search both stay 'ok' on both surfaces.
|
|
11835
|
+
* knowledgeBaseRetrieve cannot reach this state: it fetches named files rather than ranking
|
|
11836
|
+
* against a query embedding, so it has no comparison to come up empty.
|
|
11837
|
+
* 'failed' - recall did not complete: it threw, OR the retrieval repository is not wired on
|
|
11838
|
+
* this host (the guards in ChatCompletionFeatures / knowledgeBaseSearch / knowledgeBaseRetrieve
|
|
11839
|
+
* record it without anything throwing). What separates it from 'not_indexed' is the remedy,
|
|
11840
|
+
* not the tempo: fix the outage or the host wiring, never re-index content. An unwired host
|
|
11841
|
+
* reports continuously too, so "chronic" alone does not pick out 'not_indexed'.
|
|
11842
|
+
* NOT this: a model-supplied argument that is not a well-formed id. knowledgeBaseRetrieve
|
|
11843
|
+
* shape-checks `file_id` and answers a malformed one as a single-file miss ('ok'), because the
|
|
11844
|
+
* remedy is for the model to search for the right id - there is nothing for an operator to
|
|
11845
|
+
* fix. It is logged rather than counted here, so the rate stays observable without this field
|
|
11846
|
+
* reporting an outage that is not happening.
|
|
11847
|
+
* On multiple retrieval calls within one turn, merge priority is failed > not_indexed > ok >
|
|
11848
|
+
* no_lakes (see retrievalSummaryMerge.ts's mergeRetrievalSummary): a single failure is never
|
|
11849
|
+
* masked by a later success or abstain, an unsearchable corpus outranks a legitimate zero so a
|
|
11850
|
+
* success on another surface cannot erase it, and a real success is never masked by another
|
|
11851
|
+
* surface's "no lakes in scope" abstain in the same turn.
|
|
10804
11852
|
*/
|
|
10805
11853
|
outcome: z$1.enum([
|
|
10806
11854
|
"ok",
|
|
10807
11855
|
"no_lakes",
|
|
11856
|
+
"not_indexed",
|
|
10808
11857
|
"failed"
|
|
10809
|
-
]),
|
|
11858
|
+
]).optional(),
|
|
11859
|
+
/**
|
|
11860
|
+
* Whether forced retrieval was ENABLED for this turn, independent of whether it then ran.
|
|
11861
|
+
*
|
|
11862
|
+
* This is what makes the optional path measurable. `attempted` says retrieval happened;
|
|
11863
|
+
* without `mode` there is no way to ask the complementary question - of the turns where the
|
|
11864
|
+
* model was merely OFFERED the knowledge tools, how often did it choose to retrieve - because
|
|
11865
|
+
* a forced turn and an optional turn both land as `attempted: true`.
|
|
11866
|
+
*
|
|
11867
|
+
* Optional on the schema, and absence NEVER means 'optional' - it means unclassified. Two
|
|
11868
|
+
* sources, one historical and one ongoing: turns recorded before this field landed, and
|
|
11869
|
+
* agent-mode runs, which write a retrieval summary through `persistRunAsQuest` but never pass
|
|
11870
|
+
* the seed site at the `offeredTools` write in ChatCompletionProcess. So date-bounding a rollup
|
|
11871
|
+
* removes the first source but not the second; count the unclassified bucket rather than
|
|
11872
|
+
* assuming it empties.
|
|
11873
|
+
*/
|
|
11874
|
+
mode: z$1.enum(["forced", "optional"]).optional(),
|
|
11875
|
+
/**
|
|
11876
|
+
* Whether the knowledge-base when-to-retrieve guidance section actually shipped in this turn's
|
|
11877
|
+
* tool prompt. Written only on turns that were OFFERED the knowledge tool, so absence means
|
|
11878
|
+
* "not an offered turn" or "recorded before this field landed" - it never means "cleared".
|
|
11879
|
+
*
|
|
11880
|
+
* `false` is the load-bearing value here, not filler. Clearing the KnowledgeBaseRetrievalPrompt
|
|
11881
|
+
* setting is the section's only off switch, so a turn recording `false` is the CONTROL arm of
|
|
11882
|
+
* the A/B this field exists to make readable. Anything merging or folding this must preserve an
|
|
11883
|
+
* explicit `false` rather than collapse it into absent - see mergeRetrievalSummary, which uses
|
|
11884
|
+
* `??` and deliberately not `||` for that reason.
|
|
11885
|
+
*
|
|
11886
|
+
* MUST STAY IN SYNC with TWO gates, not one. ToolBuilder.buildToolPrompt emits the section iff
|
|
11887
|
+
* the tool is offered AND the guidance string is non-empty; filterByPromptMode then drops the
|
|
11888
|
+
* whole `toolPrompt` source, which no promptMode admits, so an offered tool is not sufficient.
|
|
11889
|
+
* The ChatCompletionProcess seed site conjoins all three, and hands the first two to
|
|
11890
|
+
* buildToolPrompt as the same consts, so the flag and the actual emission cannot drift.
|
|
11891
|
+
*/
|
|
11892
|
+
knowledgeBaseGuidanceInjected: z$1.boolean().optional(),
|
|
11893
|
+
/**
|
|
11894
|
+
* Why the forced arm did not run on a turn that had it enabled. Only ever set with
|
|
11895
|
+
* `mode: 'forced'`, and only for the deliberate suppressions in
|
|
11896
|
+
* ChatCompletionFeatures.getContextMessages - a forced turn that ran and failed reports that
|
|
11897
|
+
* through `outcome`, not here.
|
|
11898
|
+
*
|
|
11899
|
+
* These turns are the reason this field exists: forced retrieval is configured, a rule
|
|
11900
|
+
* suppresses it, and the model falls back to the offered tool. That is exactly the population
|
|
11901
|
+
* the per-turn routing question is about, and before this it was indistinguishable from a turn
|
|
11902
|
+
* where forced retrieval was never configured at all.
|
|
11903
|
+
*/
|
|
11904
|
+
forcedSkipReason: z$1.enum(["attached_files", "personal_corpus"]).optional(),
|
|
10810
11905
|
/** Which retrieval-capable surface(s) ran this turn, e.g. 'lake-memory', 'knowledgeBaseSearch'. */
|
|
10811
11906
|
surfaces: z$1.array(z$1.string()),
|
|
10812
11907
|
/** Lakes resolved at the moment retrieval ran, stamped point-in-time (not read live from the session). */
|
|
10813
|
-
dataLakeTags: z$1.array(z$1.string())
|
|
11908
|
+
dataLakeTags: z$1.array(z$1.string()),
|
|
11909
|
+
/**
|
|
11910
|
+
* The lake scope the turn's retrieval surfaces WOULD have searched, resolved at the seed site
|
|
11911
|
+
* whether or not any of them ran: the caller's accessible lakes narrowed to the session
|
|
11912
|
+
* (narrowLakeAccessToSession), or empty where the corpus is personal and the lake arms are
|
|
11913
|
+
* suppressed. `dataLakeTags` is the other half of the pair and answers a different question -
|
|
11914
|
+
* which lakes retrieval ACTUALLY used - so on a turn where retrieval never ran that one is empty
|
|
11915
|
+
* while this one still names whatever was in scope.
|
|
11916
|
+
*
|
|
11917
|
+
* EXISTS FOR THE OFFLINE REPLAY. `answerability` is reconstructed after the fact, and without a
|
|
11918
|
+
* recorded scope the replay had to rebuild one from the session's `retrievalTags` as they stand
|
|
11919
|
+
* at replay time - a session whose lake selection had since changed was replayed against a
|
|
11920
|
+
* corpus its turn never had, with nothing to flag it. Recording it here removes that drift for
|
|
11921
|
+
* every turn seeded after this landed; `probedAt` still discloses the content drift, which no
|
|
11922
|
+
* amount of recording can fix.
|
|
11923
|
+
*
|
|
11924
|
+
* Absence means NOT RECORDED (a turn predating this, or one whose `retrieval` was written only
|
|
11925
|
+
* by a surface rather than by the seed) - never "no lakes in scope", which is present-and-empty.
|
|
11926
|
+
* The replay must keep those apart: probing an unrecorded turn would mean inventing a scope,
|
|
11927
|
+
* which is the approximation this field exists to end.
|
|
11928
|
+
*
|
|
11929
|
+
* Only the seed writes it, so mergeRetrievalSummary carries it first-writer-wins rather than
|
|
11930
|
+
* unioning: a later surface write asserting a narrower scope must not be able to widen the
|
|
11931
|
+
* recorded one, and a union across the two would mean neither.
|
|
11932
|
+
*/
|
|
11933
|
+
lakeScope: z$1.array(z$1.string()).optional(),
|
|
11934
|
+
/**
|
|
11935
|
+
* Ids of the lakes whose `systemPrompt` was injected this turn (getAccessibleDataLakePrompts),
|
|
11936
|
+
* across every injection site (forced retrieval and the model-driven knowledge tools). NOT the
|
|
11937
|
+
* prompt text itself - that already reaches the model in the completion, and copying it here
|
|
11938
|
+
* widens exposure for nothing. Absent means no injection site ran; present-and-empty means one
|
|
11939
|
+
* ran but nothing qualified (untrusted, or an empty systemPrompt).
|
|
11940
|
+
*/
|
|
11941
|
+
injectedLakePromptIds: z$1.array(z$1.string()).optional(),
|
|
11942
|
+
/** mementoCount/mementoIds precedent: mirrors injectedLakePromptIds.length. */
|
|
11943
|
+
injectedLakePromptCount: z$1.number().optional(),
|
|
11944
|
+
/**
|
|
11945
|
+
* How much retrieved content actually reached the model this turn: `chunks` passages totalling
|
|
11946
|
+
* `chars` characters of retrieved CONTENT (headings and framing excluded, so the number means
|
|
11947
|
+
* the same thing on every surface), plus `topScore`, the best similarity among the compared
|
|
11948
|
+
* passages the reporting surface can SEE - which is not the same population on every surface.
|
|
11949
|
+
* Forced retrieval scores every chunk itself and so reports true near-misses; knowledgeBaseSearch's
|
|
11950
|
+
* semantic arm only ever sees `minScore` survivors, and reports no `topScore` at all on a starve,
|
|
11951
|
+
* so a sub-floor near-miss there is invisible rather than recorded.
|
|
11952
|
+
*
|
|
11953
|
+
* PRESENCE CONTRACT: present if and only if at least one surface COMPLETED a search this turn.
|
|
11954
|
+
* `chunks: 0` is a RECORDED STARVE - the library was searched and nothing was injected, which is
|
|
11955
|
+
* the case this field exists to make visible: without it, a forced-retrieval turn that injected
|
|
11956
|
+
* nothing is byte-identical to one that injected its whole character budget (both `outcome:
|
|
11957
|
+
* 'ok'`). Absence means the volume is UNKNOWN, which is what a turn carries when no surface
|
|
11958
|
+
* completed a search: retrieval was never attempted, nothing was in scope to search
|
|
11959
|
+
* ('no_lakes'), or the one surface that ran broke mid-flight, where a zero would be a lie.
|
|
11960
|
+
* A surface that completed but CANNOT know the turn's passage volume also stays silent rather
|
|
11961
|
+
* than claiming a zero - knowledgeBaseSearch's keyword arm on a hit is the case: it injects
|
|
11962
|
+
* file metadata and hands the model retrieve_knowledge_content, which injects the text and
|
|
11963
|
+
* reports no volume, so its zero would survive the merge as a starve that did not happen.
|
|
11964
|
+
*
|
|
11965
|
+
* KNOWN HOLE in that rule, while retrieve_knowledge_content stays uninstrumented: a recorded zero
|
|
11966
|
+
* is not PROOF of a starve. Forced retrieval and the knowledge tools are not mutually exclusive
|
|
11967
|
+
* (ChatCompletionProcess seeds on `forcedRetrievalEnabled || knowledgeToolOffered`), so the forced
|
|
11968
|
+
* arm can complete empty, write its honest zero, and the model can then ground the same turn
|
|
11969
|
+
* through retrieve_knowledge_content, which contributes no volume to oppose it. The zero is
|
|
11970
|
+
* per-surface-truthful and turn-level-misleading. Any rollup counting starves should treat a zero
|
|
11971
|
+
* as "nothing was injected by a surface that reports volume" and, until that tool reports its own,
|
|
11972
|
+
* cross-check `functionCalls` before calling the turn ungrounded.
|
|
11973
|
+
*
|
|
11974
|
+
* Per SURFACE, not per turn: a surface that breaks contributes nothing while a surface that
|
|
11975
|
+
* completed alongside it still reports its own volume, so a turn CAN read 'failed' next to a
|
|
11976
|
+
* recorded zero. That pairing means "one surface broke, and everything that did finish injected
|
|
11977
|
+
* nothing" - which is exactly what a reader needs, and strictly more than the outcome alone.
|
|
11978
|
+
*
|
|
11979
|
+
* SUMMED across surfaces, so this field and `outcome` can legitimately disagree in tone on a
|
|
11980
|
+
* multi-surface turn: forced retrieval grounding on 12 passages while knowledgeBaseSearch throws
|
|
11981
|
+
* gives `outcome: 'failed'` alongside `chunks: 12`. That is correct - `outcome` is worst-of,
|
|
11982
|
+
* `injected` is sum-of-completions.
|
|
11983
|
+
*
|
|
11984
|
+
* `topScore` is optional because only cosine-similarity surfaces have one to report. Lake
|
|
11985
|
+
* memory's belief `relevance` is a different scale and forced retrieval's pre-scan value is a
|
|
11986
|
+
* -1 sentinel; neither is ever written here, because a `max` across mixed scales, or against a
|
|
11987
|
+
* sentinel, is a number that reads as a similarity and is not one.
|
|
11988
|
+
*
|
|
11989
|
+
* Date-bound any rollup, the same caveat `mode` documents on itself: turns recorded before this
|
|
11990
|
+
* landed carry no volume, and no backfill is possible - the volume of a past turn is gone.
|
|
11991
|
+
*
|
|
11992
|
+
* `preRelativeFloorCandidates` and `postRelativeFloorCandidates` are the ONE pair here that is
|
|
11993
|
+
* not "what reached the model": `ranked.length` and `scored.length` in KnowledgeRetrievalFeature
|
|
11994
|
+
* - the candidates left after the absolute similarity floor, and after the relative floor
|
|
11995
|
+
* trims them. `chunks` is what survived the char budget on top of that, so the three
|
|
11996
|
+
* numbers bracket two independent trimmers:
|
|
11997
|
+
*
|
|
11998
|
+
* pre -> [relative floor] -> post -> [char budget] -> chunks
|
|
11999
|
+
*
|
|
12000
|
+
* They exist so a low `chunks` is diagnosable - a small corpus and a floor that trimmed a large
|
|
12001
|
+
* pool end in the same `chunks`. `pre - post` is the floor's own effect and nothing else;
|
|
12002
|
+
* `pre - chunks` is NOT, because the budget trims the same walk. Both optional: only forced
|
|
12003
|
+
* retrieval computes a ranked pool, a surface without one (lake memory, the knowledge tools)
|
|
12004
|
+
* never writes either, and absence must not read as zero candidates. SUMMED like `chunks`, with
|
|
12005
|
+
* the same absent-is-not-zero handling as `topScore`.
|
|
12006
|
+
*
|
|
12007
|
+
* COMPARE THE PAIR ONLY TO ITSELF, never to `chunks`, unless `surfaces` is forced retrieval
|
|
12008
|
+
* alone. `chunks` and `chars` sum across ALL surfaces while this pair is forced-only, so a mixed
|
|
12009
|
+
* turn can store `chunks` above `pre` - inverting the relationship the pair exposes. Lake memory
|
|
12010
|
+
* is the common case, not the exotic one: it is enabled inside the same forced-retrieval gate,
|
|
12011
|
+
* so on a lake-memory lake it writes on nearly every forced turn. Its chunks can be backed out
|
|
12012
|
+
* via `context.lakeMemory.beliefCount` (approximately - that count is pre-sanitization); its
|
|
12013
|
+
* CHARS land only inside the shared sum, with no per-surface field to subtract them back out, so
|
|
12014
|
+
* `chars` cannot be decontaminated at all. `pre - post` needs neither, which is the point of
|
|
12015
|
+
* storing both.
|
|
12016
|
+
*
|
|
12017
|
+
* BOTH SATURATE, so `pre` counts what the SCAN REACHED, not what the corpus holds: `pool` is
|
|
12018
|
+
* truncated in-scan at FORCED_RETRIEVAL_MAX_SCORED_CHUNKS (256), over a scan itself bounded by
|
|
12019
|
+
* FORCED_RETRIEVAL_MAX_SCANNED_CHUNKS (4000) across FORCED_RETRIEVAL_MAX_CANDIDATE_FILES (100).
|
|
12020
|
+
* 2000 qualifying chunks and 300 both record 256; above the cap a rollup is a plateau.
|
|
12021
|
+
*/
|
|
12022
|
+
injected: z$1.object({
|
|
12023
|
+
chunks: z$1.number(),
|
|
12024
|
+
chars: z$1.number(),
|
|
12025
|
+
topScore: z$1.number().optional(),
|
|
12026
|
+
preRelativeFloorCandidates: z$1.number().optional(),
|
|
12027
|
+
postRelativeFloorCandidates: z$1.number().optional()
|
|
12028
|
+
}).optional(),
|
|
12029
|
+
/**
|
|
12030
|
+
* Could the corpus in scope have answered this turn, whether or not the model went looking?
|
|
12031
|
+
*
|
|
12032
|
+
* The denominator the optional-path retrieval rate has always been missing (#1394). A rate of
|
|
12033
|
+
* "the model retrieved on 20% of offered turns" cannot say whether the other 80% were misses or
|
|
12034
|
+
* turns with nothing to find, and the two argue for opposite things: the first for routing work,
|
|
12035
|
+
* the second for leaving the optional path alone. Crossing this field with the rate separates
|
|
12036
|
+
* them.
|
|
12037
|
+
*
|
|
12038
|
+
* THE ONLY FIELD IN THIS BLOCK NOT WRITTEN BY THE TURN. Every sibling is stamped point-in-time
|
|
12039
|
+
* while the turn runs; this one is written afterwards by an offline replay
|
|
12040
|
+
* (packages/scripts/retrieval/answerability-replay.ts) that re-scores the recorded prompt against
|
|
12041
|
+
* the corpus. That is deliberate - the population it exists to measure is the turns where
|
|
12042
|
+
* retrieval did NOT run, so computing it live would mean adding a full brute-force chunk scan
|
|
12043
|
+
* (ChatCompletionFeatures' forced path, which has no ANN index) to exactly the turns that pay
|
|
12044
|
+
* nothing for retrieval today. The measurement is not worth that latency on live traffic.
|
|
12045
|
+
*
|
|
12046
|
+
* BEING A RECONSTRUCTION, IT CARRIES DRIFTS THE OTHER FIELDS DO NOT:
|
|
12047
|
+
* 1. Corpus CONTENT moves. A document added or reindexed between the turn and the replay is
|
|
12048
|
+
* scored as though it had been there. `probedAt` discloses the gap; a replay run long after
|
|
12049
|
+
* the window is weak evidence, not strong.
|
|
12050
|
+
* 2. Corpus SCOPE no longer drifts: the seed records the turn's resolved scope in `lakeScope`
|
|
12051
|
+
* and the replay probes that, so a session whose lake selection has since changed is still
|
|
12052
|
+
* scored against the lakes its turn actually had. The cost is coverage rather than accuracy -
|
|
12053
|
+
* a turn with no recorded scope is skipped instead of approximated, so every turn predating
|
|
12054
|
+
* the field is outside the measurement.
|
|
12055
|
+
* 3. The QUESTION can move out from under it. The probe is keyed to the quest, not to the
|
|
12056
|
+
* prompt text it scored, so a turn whose prompt is later rewritten in place keeps a probe
|
|
12057
|
+
* describing the question it used to ask. mergeRetrievalSummary preserves the probe across
|
|
12058
|
+
* a runtime write deliberately - dropping it would erase the backfill - so nothing
|
|
12059
|
+
* invalidates a stale one. Re-run the replay with --force over a window whose turns were
|
|
12060
|
+
* edited.
|
|
12061
|
+
*
|
|
12062
|
+
* RAW SCORE, NOT A VERDICT, so the cutoff lives in the reader. summarizeOptionalPathRetrieval
|
|
12063
|
+
* applies it at fold time, which lets the same replay be re-thresholded without re-running -
|
|
12064
|
+
* the point of storing the number, given the two live floors disagree by construction (forced
|
|
12065
|
+
* retrieval's absolute default is 0.75, the knowledge tool's is 0).
|
|
12066
|
+
*
|
|
12067
|
+
* `topScore` is the same raw cosine scale as `injected.topScore` and comparable to it. It is NOT
|
|
12068
|
+
* comparable to lake memory's belief relevance, for the reason `injected` documents at length.
|
|
12069
|
+
*
|
|
12070
|
+
* `scanTruncated` inherits forced retrieval's saturation: the replay bounds its scan the same
|
|
12071
|
+
* way, so a low `topScore` on a truncated scan is not proof the corpus lacked an answer - it is
|
|
12072
|
+
* proof the part that was scanned did. Treat those turns as unknown rather than as negatives.
|
|
12073
|
+
*
|
|
12074
|
+
* Absence means NOT PROBED - never "not answerable". Every turn predating the replay, and every
|
|
12075
|
+
* turn the replay skipped or failed on, is absent, so a fold must keep it as its own arm rather
|
|
12076
|
+
* than letting it fall in with the negatives.
|
|
12077
|
+
*/
|
|
12078
|
+
answerability: z$1.object({
|
|
12079
|
+
/** Best cosine the replay found across the reconstructed corpus. */
|
|
12080
|
+
topScore: z$1.number(),
|
|
12081
|
+
/** Chunks at or above `floor`. Separates "one lucky match" from "a rich seam". */
|
|
12082
|
+
candidatesAboveFloor: z$1.number(),
|
|
12083
|
+
/** The absolute floor the replay counted `candidatesAboveFloor` against, as a fraction. */
|
|
12084
|
+
floor: z$1.number(),
|
|
12085
|
+
/** The scan hit its chunk ceiling, so `topScore` is a floor on the true best, not the best. */
|
|
12086
|
+
scanTruncated: z$1.boolean(),
|
|
12087
|
+
/** When the replay ran, NOT when the turn ran - the disclosure for content drift above. */
|
|
12088
|
+
probedAt: JsonSafeDate
|
|
12089
|
+
}).optional(),
|
|
12090
|
+
/**
|
|
12091
|
+
* Which of this turn's injected lake prompt ids were BOTH in the session's pre-authorized (manage-
|
|
12092
|
+
* but-not-member admission) set AND injected on this turn - see unionPreauthorizedLakeAccess and
|
|
12093
|
+
* pages/api/sessions/create.ts. A subset of injectedLakePromptIds, never a superset. Narrows the
|
|
12094
|
+
* session's static `preauthorizedLakeIds` (what was ADMITTED) to what a given turn actually used.
|
|
12095
|
+
*
|
|
12096
|
+
* MEMBERSHIP, NOT CAUSATION. An admitted lake the caller could already reach - its creator, or a
|
|
12097
|
+
* member of its org - injects through the ordinary trust arm and is listed here all the same, so a
|
|
12098
|
+
* non-empty value does not prove the admission is what made the injection possible. Absent means no
|
|
12099
|
+
* admitted id was among this turn's injections, including every turn on a session with none.
|
|
12100
|
+
*/
|
|
12101
|
+
preauthorizedLakeIdsUsed: z$1.array(z$1.string()).optional(),
|
|
12102
|
+
/**
|
|
12103
|
+
* Which of this turn's injected lake prompt ids the caller holds an owner/curator GRANT on - the
|
|
12104
|
+
* per-arm sibling of `preauthorizedLakeIdsUsed`, and the reason both exist: `injectedLakePromptIds`
|
|
12105
|
+
* records THAT a lake's systemPrompt entered the turn, these two record WHICH ARM admitted it.
|
|
12106
|
+
* A subset of injectedLakePromptIds, never a superset. Derived at both injection sites via
|
|
12107
|
+
* grantedLakeIdsUsedFor.
|
|
12108
|
+
*
|
|
12109
|
+
* THE ARM WORTH NAMING SEPARATELY: the grant arm (#2495) is the only one that can cross an org
|
|
12110
|
+
* boundary - `grantLakeAccess` can hand a CURATOR grant to an arbitrary cross-tenant user, and
|
|
12111
|
+
* that grant carries injection trust. The creator and org arms cannot reach past one org, and a
|
|
12112
|
+
* reader grant is excluded permanently, so a lake listed here is the case an operator auditing
|
|
12113
|
+
* cross-tenant prompt influence is actually looking for.
|
|
12114
|
+
*
|
|
12115
|
+
* MEMBERSHIP, NOT CAUSATION, the same caveat the field above carries: a granted lake its holder
|
|
12116
|
+
* could already reach (they created it, or they are in its org) injects through the ordinary
|
|
12117
|
+
* trust arm and is listed here anyway, so a non-empty value does not prove the grant is what
|
|
12118
|
+
* made the injection possible. The two fields OVERLAP for that reason - a lake can appear in
|
|
12119
|
+
* both - so they are not a partition of injectedLakePromptIds and must not be counted as one.
|
|
12120
|
+
*
|
|
12121
|
+
* Absent means no injected id was grant-reached, including every turn where the grant read
|
|
12122
|
+
* FAILED (it is fail-quiet by design - telemetry must not drop the injection it records), so
|
|
12123
|
+
* absence is weaker evidence than presence. Date-bound any rollup: turns predating this field
|
|
12124
|
+
* carry nothing, and no backfill is possible - a past turn's grant rows have moved on.
|
|
12125
|
+
*/
|
|
12126
|
+
grantedLakeIdsUsed: z$1.array(z$1.string()).optional()
|
|
12127
|
+
});
|
|
12128
|
+
/**
|
|
12129
|
+
* Why a grounded turn's library scan stopped short of the whole library.
|
|
12130
|
+
*
|
|
12131
|
+
* Written ONLY on a partially-covered turn (reportCoverage returns early otherwise), so presence
|
|
12132
|
+
* means "partial" and `partial` is always true - the flag is explicit anyway because a reader
|
|
12133
|
+
* checking `retrievalCoverage.partial` should not have to know that absence is the other half of
|
|
12134
|
+
* the contract.
|
|
12135
|
+
*
|
|
12136
|
+
* Single producer (ChatCompletionFeatures.reportCoverage), which is why - unlike `warnings`,
|
|
12137
|
+
* `citables` and `retrieval` - this field needs no merge case in applyQuestStatusChanges: a
|
|
12138
|
+
* later tool-arm write that omits it is preserved by the one-level spread.
|
|
12139
|
+
*
|
|
12140
|
+
* `reasons` is the same diagnostic prose the warnings entry interpolates. It is shown to the
|
|
12141
|
+
* reader behind a disclosure rather than in the banner body, because only some reasons are
|
|
12142
|
+
* actionable (a document mid-reindex returns on its own; a per-turn chunk budget does not).
|
|
12143
|
+
*/
|
|
12144
|
+
const RetrievalCoverageSchema = z$1.object({
|
|
12145
|
+
/** Always true - see the presence contract above. */
|
|
12146
|
+
partial: z$1.boolean(),
|
|
12147
|
+
/** One entry per distinct cause, e.g. a candidate cap, a scan budget, an embedding mismatch. */
|
|
12148
|
+
reasons: z$1.array(z$1.string())
|
|
10814
12149
|
});
|
|
10815
12150
|
z$1.object({
|
|
10816
12151
|
model: PromptMetaModelSchema.optional(),
|
|
@@ -10820,6 +12155,9 @@ z$1.object({
|
|
|
10820
12155
|
* deliberately: applyQuestStatusChanges does a one-level spread merge, so a field nested under
|
|
10821
12156
|
* `context` would be replaced wholesale by any tool-arm write instead of merging. */
|
|
10822
12157
|
retrieval: RetrievalSummarySchema.optional(),
|
|
12158
|
+
/** Partial-grounding-coverage detail - see RetrievalCoverageSchema. Top-level for the same
|
|
12159
|
+
* one-level-spread-merge reason as `retrieval` above. */
|
|
12160
|
+
retrievalCoverage: RetrievalCoverageSchema.optional(),
|
|
10823
12161
|
functionCalls: z$1.array(PromptMetaFunctionCallSchema).optional(),
|
|
10824
12162
|
/**
|
|
10825
12163
|
* Names of the tools actually offered to the model this turn - the output of `buildTools`
|
|
@@ -11513,6 +12851,7 @@ z.object({
|
|
|
11513
12851
|
requiredUserTag: z.union([z.literal(""), z.string().min(1).max(100)]).optional(),
|
|
11514
12852
|
requiredEntitlement: z.union([z.literal(""), z.string().min(3).max(100).refine((s) => s.includes(":") && s.split(":").every((part) => part.length > 0), "Entitlement key must be namespaced with non-empty parts (e.g. \"product:pro\")")]).optional(),
|
|
11515
12853
|
auditQueryTextEnabled: z.boolean().optional(),
|
|
12854
|
+
lakeMemoryEnabled: z.boolean().optional(),
|
|
11516
12855
|
requiredPassageTokenTarget: z.number().int().min(64).max(OVERSIZED_PASSAGE_TOKEN_THRESHOLD).nullable().optional()
|
|
11517
12856
|
});
|
|
11518
12857
|
z.object({
|
|
@@ -11587,6 +12926,7 @@ tags: z.array(TaxonomyTagInput).max(100) });
|
|
|
11587
12926
|
z.object({
|
|
11588
12927
|
/** User description of the data (helps the AI) */
|
|
11589
12928
|
context: z.string().max(2e3).optional() });
|
|
12929
|
+
z.object({ tags: z.array(z.string().min(1).max(130).regex(/^[^\r\n]*$/)).max(100) });
|
|
11590
12930
|
z.object({ hashes: z.array(z.string().regex(sha256Regex)).min(1).max(500) });
|
|
11591
12931
|
const SyncDeltaFileEntry = z.object({
|
|
11592
12932
|
relativePath: z.string(),
|
|
@@ -11618,55 +12958,6 @@ z.object({
|
|
|
11618
12958
|
skip: z.array(z.string())
|
|
11619
12959
|
})
|
|
11620
12960
|
});
|
|
11621
|
-
z$1.enum([
|
|
11622
|
-
"inject",
|
|
11623
|
-
"auto-fire",
|
|
11624
|
-
"hidden"
|
|
11625
|
-
]);
|
|
11626
|
-
/**
|
|
11627
|
-
* Modes acceptable at AUTHORING time. 'hidden' is intentionally excluded until
|
|
11628
|
-
* the host has true hidden-send support - accepting it would persist a value
|
|
11629
|
-
* that silently behaves as 'auto-fire' (a surprising downgrade). It stays in
|
|
11630
|
-
* ExecutionModeSchema/the stored enum for forward-compat.
|
|
11631
|
-
*/
|
|
11632
|
-
const AuthorableExecutionModeSchema = z$1.enum(["inject", "auto-fire"]);
|
|
11633
|
-
/**
|
|
11634
|
-
* Tools a prompt may require - constrained to the host's closed tool set, MINUS
|
|
11635
|
-
* integration-gated tools that act on the caller's own credentials/account. A
|
|
11636
|
-
* shared system prompt must not be able to inject e.g. blog-publishing into a
|
|
11637
|
-
* non-author's session via requiredTools. (Per-user entitlement of the remaining
|
|
11638
|
-
* tools is still the chat pipeline's responsibility - see follow-up note in the
|
|
11639
|
-
* briefcase blueprint; this allowlist is the storage-layer floor.)
|
|
11640
|
-
*/
|
|
11641
|
-
const BRIEFCASE_DISALLOWED_TOOLS = [
|
|
11642
|
-
"blog_publish",
|
|
11643
|
-
"blog_edit",
|
|
11644
|
-
"blog_draft"
|
|
11645
|
-
];
|
|
11646
|
-
const BriefcaseRequiredToolsSchema = z$1.array(b4mLLMTools.refine((t) => !BRIEFCASE_DISALLOWED_TOOLS.includes(t), "This tool is not permitted in a briefcase prompt")).max(16);
|
|
11647
|
-
z$1.string().regex(/^[a-f0-9]{24}$/i, "Invalid prompt id");
|
|
11648
|
-
const PROMPT_TEXT_MAX = 16e3;
|
|
11649
|
-
const TAGS_MAX = 20;
|
|
11650
|
-
z$1.object({
|
|
11651
|
-
type: z$1.string().min(1).max(100),
|
|
11652
|
-
name: z$1.string().min(1).max(200),
|
|
11653
|
-
description: z$1.string().max(500).optional(),
|
|
11654
|
-
promptText: z$1.string().min(1).max(PROMPT_TEXT_MAX),
|
|
11655
|
-
tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
|
|
11656
|
-
executionMode: AuthorableExecutionModeSchema.optional(),
|
|
11657
|
-
requiredTools: BriefcaseRequiredToolsSchema.optional()
|
|
11658
|
-
}).partial();
|
|
11659
|
-
/**
|
|
11660
|
-
* One catalog sub-query. Exactly one selector is used, in precedence order:
|
|
11661
|
-
* `personal` (resolved to the caller server-side) > `tags` > `type`.
|
|
11662
|
-
*/
|
|
11663
|
-
const PromptBatchQuerySchema = z$1.object({
|
|
11664
|
-
key: z$1.string().min(1).max(100),
|
|
11665
|
-
tags: z$1.array(z$1.string().min(1).max(50)).max(TAGS_MAX).optional(),
|
|
11666
|
-
type: z$1.string().max(100).optional(),
|
|
11667
|
-
personal: z$1.boolean().optional()
|
|
11668
|
-
});
|
|
11669
|
-
z$1.object({ queries: z$1.array(PromptBatchQuerySchema).min(1).max(32).refine((qs) => new Set(qs.map((q) => q.key)).size === qs.length, { message: "Batch query keys must be unique" }) });
|
|
11670
12961
|
z$1.string().regex(/^[a-f0-9]{24}$/i, "Invalid template id");
|
|
11671
12962
|
/**
|
|
11672
12963
|
* The bound model. Reuses the legacy-remap preprocess so a template saved under
|
|
@@ -11969,6 +13260,31 @@ z$1.object({
|
|
|
11969
13260
|
"grounded",
|
|
11970
13261
|
"surface"
|
|
11971
13262
|
]).optional(),
|
|
13263
|
+
/**
|
|
13264
|
+
* Suppress OUR server-side auto-offers without entering a promptMode. Exists because promptMode
|
|
13265
|
+
* was the only switch for the offer and it also strips every authored prompt, so no caller could
|
|
13266
|
+
* have an arm that went unoffered AND kept the abstention licence.
|
|
13267
|
+
*
|
|
13268
|
+
* Gates the three auto-add sites (the knowledge offer in resolveEnabledTools, the navigate_view
|
|
13269
|
+
* auto-add, the blog/skill gate), unioned with `Boolean(promptMode)` by
|
|
13270
|
+
* resolveSkipAutoOffers. A force-on, not an override: `false` under a promptMode still suppresses.
|
|
13271
|
+
* Withholding navigate_view also drops the viewRegistry system block, which only describes it.
|
|
13272
|
+
*
|
|
13273
|
+
* Withholds the OFFER, not knowledge: `session.forceKnowledgeRetrieval` is untouched, and an
|
|
13274
|
+
* already-attached corpus is inlined rather than deferred to the tool. An arm that must see no
|
|
13275
|
+
* knowledge at all also needs a session with no attachments and forced retrieval off.
|
|
13276
|
+
*/
|
|
13277
|
+
skipAutoOffers: z$1.boolean().optional(),
|
|
13278
|
+
/**
|
|
13279
|
+
* Caller-supplied system-prompt text. Rendered as a defended, deference-postured block
|
|
13280
|
+
* appended last in the system-prompt stack. Reached by both POST /api/chat and /api/ai/llm.
|
|
13281
|
+
*
|
|
13282
|
+
* This cap is the universal backstop, not a duplicate of a route check: the parse that opens
|
|
13283
|
+
* invoke() runs outside any try and before a quest row is written, so it holds for every caller
|
|
13284
|
+
* including ones that pass through no route schema. Do not drop it on the assumption that
|
|
13285
|
+
* whoever called validated first.
|
|
13286
|
+
*/
|
|
13287
|
+
systemPrompt: z$1.string().max(PROMPT_TEXT_MAX).optional(),
|
|
11972
13288
|
/** Whether Mementos is enabled */
|
|
11973
13289
|
enableMementos: z$1.boolean().optional(),
|
|
11974
13290
|
/** Whether Artifacts is enabled */
|
|
@@ -12942,7 +14258,7 @@ Array.from(new Set([
|
|
|
12942
14258
|
id: "opti.root",
|
|
12943
14259
|
section: "opti",
|
|
12944
14260
|
label: "OptiHashi Home",
|
|
12945
|
-
description: "The OptiHashi Optimizer landing page showing
|
|
14261
|
+
description: "The OptiHashi Optimizer landing page showing the pattern family cards",
|
|
12946
14262
|
navigationType: "route",
|
|
12947
14263
|
target: "/opti",
|
|
12948
14264
|
keywords: [
|
|
@@ -13843,6 +15159,66 @@ Array.from(new Set([
|
|
|
13843
15159
|
const [, top] = v.target.split("/");
|
|
13844
15160
|
return `/${top}`;
|
|
13845
15161
|
})));
|
|
15162
|
+
function getHeader(headers, name) {
|
|
15163
|
+
if (!headers || typeof headers !== "object") return null;
|
|
15164
|
+
if (typeof headers.get === "function") {
|
|
15165
|
+
const value = headers.get(name);
|
|
15166
|
+
return typeof value === "string" ? value : null;
|
|
15167
|
+
}
|
|
15168
|
+
const value = headers[name] ?? headers[name.toLowerCase()];
|
|
15169
|
+
return typeof value === "string" ? value : null;
|
|
15170
|
+
}
|
|
15171
|
+
function parseCount(value) {
|
|
15172
|
+
if (value === null) return null;
|
|
15173
|
+
const trimmed = value.trim();
|
|
15174
|
+
if (!trimmed) return null;
|
|
15175
|
+
const parsed = Number(trimmed);
|
|
15176
|
+
return Number.isFinite(parsed) ? parsed : null;
|
|
15177
|
+
}
|
|
15178
|
+
const UNIT_MS = {
|
|
15179
|
+
ms: 1,
|
|
15180
|
+
s: 1e3,
|
|
15181
|
+
m: 6e4,
|
|
15182
|
+
h: 36e5
|
|
15183
|
+
};
|
|
15184
|
+
const DURATION_PART = /(\d+(?:\.\d+)?)(ms|h|m|s)/g;
|
|
15185
|
+
/**
|
|
15186
|
+
* Parse a Go-style duration ("6ms", "0s", "1m30s", "1h2m3s") to milliseconds.
|
|
15187
|
+
*
|
|
15188
|
+
* Exported for its own tests: it is the part of this module that can be wrong in a way the
|
|
15189
|
+
* numbers still look plausible.
|
|
15190
|
+
*/
|
|
15191
|
+
function parseDurationMs(value) {
|
|
15192
|
+
if (typeof value !== "string") return null;
|
|
15193
|
+
const trimmed = value.trim();
|
|
15194
|
+
if (!trimmed) return null;
|
|
15195
|
+
DURATION_PART.lastIndex = 0;
|
|
15196
|
+
let total = 0;
|
|
15197
|
+
let matched = 0;
|
|
15198
|
+
let consumed = 0;
|
|
15199
|
+
for (const part of trimmed.matchAll(DURATION_PART)) {
|
|
15200
|
+
total += Number(part[1]) * UNIT_MS[part[2]];
|
|
15201
|
+
consumed += part[0].length;
|
|
15202
|
+
matched += 1;
|
|
15203
|
+
}
|
|
15204
|
+
if (matched === 0 || consumed !== trimmed.length) return null;
|
|
15205
|
+
return total;
|
|
15206
|
+
}
|
|
15207
|
+
/** Read both rate-limit dimensions off a provider response. */
|
|
15208
|
+
function parseEmbeddingRateLimitHeaders(headers) {
|
|
15209
|
+
return {
|
|
15210
|
+
limitTokens: parseCount(getHeader(headers, "x-ratelimit-limit-tokens")),
|
|
15211
|
+
limitRequests: parseCount(getHeader(headers, "x-ratelimit-limit-requests")),
|
|
15212
|
+
remainingTokens: parseCount(getHeader(headers, "x-ratelimit-remaining-tokens")),
|
|
15213
|
+
remainingRequests: parseCount(getHeader(headers, "x-ratelimit-remaining-requests")),
|
|
15214
|
+
resetTokensMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-tokens")),
|
|
15215
|
+
resetRequestsMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-requests"))
|
|
15216
|
+
};
|
|
15217
|
+
}
|
|
15218
|
+
/** True when the provider reported at least one usable ceiling. */
|
|
15219
|
+
function hasUsableLimits(snapshot) {
|
|
15220
|
+
return snapshot.limitTokens !== null || snapshot.limitRequests !== null;
|
|
15221
|
+
}
|
|
13846
15222
|
dayjs.extend(utc);
|
|
13847
15223
|
dayjs.extend(timezone);
|
|
13848
15224
|
dayjs.extend(relativeTime);
|
|
@@ -13911,16 +15287,34 @@ function isUserInitiatedAbort(error, userSignal) {
|
|
|
13911
15287
|
return isAbortError && !userSignal;
|
|
13912
15288
|
}
|
|
13913
15289
|
/**
|
|
13914
|
-
*
|
|
15290
|
+
* A Retry-After hint is only useful if it asks us to wait. `Retry-After: 0`, a negative value, or an
|
|
15291
|
+
* HTTP date that has already passed all carry no timing information - and every `withRetry` in this
|
|
15292
|
+
* repo treats a non-null hint as authoritative *over* its exponential backoff, so returning 0 does
|
|
15293
|
+
* not mean "wait a moment", it means "abandon the backoff entirely and retry immediately".
|
|
15294
|
+
*
|
|
15295
|
+
* That inverts the retry budget exactly when it matters: a server sends Retry-After when it is
|
|
15296
|
+
* already struggling, so honouring a zero turns the remaining attempts into an instant burst against
|
|
15297
|
+
* a service asking for room. Null instead, so the caller falls through to its own backoff.
|
|
15298
|
+
*
|
|
15299
|
+
* Exported because the rule, not the code, is the thing worth sharing: `Retry-After` is parsed in
|
|
15300
|
+
* more than one package (fab-pipeline's `getOpenSearchRetryAfterMs`), and a four-token predicate
|
|
15301
|
+
* copied around is a rule that drifts. Feed it a delay already converted to ms.
|
|
15302
|
+
*/
|
|
15303
|
+
function retryAfterHintOrNull(ms) {
|
|
15304
|
+
return ms > 0 ? ms : null;
|
|
15305
|
+
}
|
|
15306
|
+
/**
|
|
15307
|
+
* Extract retry delay from error response (e.g., Retry-After header). Returns null when the header is
|
|
15308
|
+
* absent, unparseable, or does not ask us to wait - see retryAfterHintOrNull.
|
|
13915
15309
|
*/
|
|
13916
15310
|
function getRetryAfterMs(error) {
|
|
13917
15311
|
if (!isAxiosError(error)) return null;
|
|
13918
15312
|
const retryAfter = error.response?.headers?.["retry-after"];
|
|
13919
15313
|
if (!retryAfter) return null;
|
|
13920
15314
|
const seconds = parseInt(retryAfter, 10);
|
|
13921
|
-
if (!isNaN(seconds)) return seconds * 1e3;
|
|
15315
|
+
if (!isNaN(seconds)) return retryAfterHintOrNull(seconds * 1e3);
|
|
13922
15316
|
const date = Date.parse(retryAfter);
|
|
13923
|
-
if (!isNaN(date)) return
|
|
15317
|
+
if (!isNaN(date)) return retryAfterHintOrNull(date - Date.now());
|
|
13924
15318
|
return null;
|
|
13925
15319
|
}
|
|
13926
15320
|
/**
|
|
@@ -15335,4 +16729,4 @@ var ConfigStore = class {
|
|
|
15335
16729
|
}
|
|
15336
16730
|
};
|
|
15337
16731
|
//#endregion
|
|
15338
|
-
export {
|
|
16732
|
+
export { SupportedFabFileMimeTypes as $, selfClaimedActorKindSchema as $t, HTTPError as A, isPlaceholderApiKey as At, OPENAI_GPT_IMAGE_1_IMAGE_SIZES as B, reservationOutputTokens as Bt, DEFAULT_MUSIC_MODEL_ID as C, isGPTImageModel as Ct, FIXED_TEMPERATURE_MODELS as D, isMediaModelType as Dt, FIELD_GROUP_OF as E, isImageServeable as Et, MODEL_INFO_FIELD_GROUP_OF as F, isUserInitiatedAbort as Ft, PermissionDeniedError as G, toModelRecord as Gt, OllamaEmbeddingModel as H, secureParameters as Ht, McpServerName as I, isZodError as It, REFUSAL_FALLBACK_MODELS as J, withRetry as Jt, REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS as K, usdToCredits as Kt, ModelBackend as L, mapMimeTypeToArtifactType as Lt, IMAGE_SIZE_CONSTRAINTS as M, isRetryableError as Mt, ImageModels as N, isSupportedFabFileMimeType as Nt, FORMAT_PROMPT_TEMPLATE as O, isModelAccessible as Ot, InternalServerError as P, isUnlimitedHistory as Pt, SpeechToTextModels as Q, actorKindSchema as Qt, NO_TEMPERATURE_MODELS as R, obfuscateApiKey as Rt, CorruptedFileError as S, isGPTImage2Model as St, DEGENERATE_FINISH_REASON as T, isImageAttachment as Tt, OpenAIEmbeddingModel as U, settingsMap as Ut, OPENAI_GPT_IMAGE_2_IMAGE_SIZES as V, resolveHistoryFetchLimit as Vt, PROMPT_TEXT_MAX as W, toModelInfo as Wt, REVIEW_GATE_STATUS_VALUES as X, actorColorIndex as Xt, RESPONSES_API_TOOL_MODELS as Y, ACTOR_COLOR_SLOTS as Yt, SUBQUEST_STATUS_VALUES as Z, actorKindMarker as Zt, BadRequestError as _, isAudioMimeType as _t, getCreditsUrl as a, VideoModels as at, CREDIT_DEDUCT_TRANSACTION_TYPES as b, isEarlyStop as bt, requireApiUrl as c, applyModelPriceCatalog as ct, AGENT_QUEST_MANIFEST as d, defaultEmbeddingModelForEnv as dt, buildRateLimitLogEntry as en, TTS_MAX_INPUT_CHARS as et, AGENT_QUEST_MCP_URI as f, getMcpProviderMetadata as ft, BFL_SAFETY_TOLERANCE as g, hasUsableLimits as gt, BEDROCK_NO_PROMPT_CACHING_MODELS as h, hasKeylessCloudEmbedder as ht, LOCAL_DEV_URL as i, VIDEO_SIZE_CONSTRAINTS as it, HttpStatus as j, isRenderableModelType as jt, ForbiddenError as k, isModelDeprecated as kt, resolveApiEndpoint as l, calculateRetryDelay as lt, ApiKeyType as m, getRetryAfterMs as mt, logger as n, isNearLimit as nn, UnauthorizedError as nt, getEnvironmentName as o, VoyageAIEmbeddingModel as ot, ARTIFACT_ATTRS_PATTERN as p, getQuestErrorCode as pt, REASONING_SUPPORTED_MODELS as q, usdToCreditsStochastic as qt, ApiEndpointUnconfiguredError as r, parseRateLimitHeaders as rn, UnprocessableEntityError as rt, parseApiUrl as s, WORK_ITEM_STATUSES as st, ConfigStore as t, extractSnippetMeta as tn, TooManyRequestsError as tt, AGENT_QUEST_ID as u, dayjsConfig_default as ut, BedrockEmbeddingModel as v, isChunkRebuildPending as vt, DEFAULT_UNKNOWN_CONTEXT_WINDOW as w, isGeminiModelId as wt, ChatModels as x, isFieldGroup as xt, CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS as y, isChunkStalledFile as yt, NotFoundError as z, parseEmbeddingRateLimitHeaders as zt };
|